From cbd7299efa3d73e39e04754251aa7b3ebd5c2a30 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 02:11:56 -0700 Subject: [PATCH 1/7] [None][fix] Stabilize Qwen3 LoRA tests with KV cache manager v2 Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- tests/integration/test_lists/waives.txt | 1 - .../tests_lora_modules/test_qwen3_sanity.py | 40 +++++++++++++++++-- 2 files changed, 36 insertions(+), 5 deletions(-) diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index d4f2d2e24421..caff39ff9dc9 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -266,7 +266,6 @@ unittest/_torch/modeling/test_modeling_nemotron_h.py::test_nemotron_h_breakable_ unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py::test_nemotron_nano_v2_vl_image_batch_equivalence SKIP (https://nvbugs/6758946) unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py::test_nemotron_nano_v2_vl_video_batch_equivalence SKIP (https://nvbugs/6625695) unittest/_torch/modules/tests_lora_modules/test_nemotron_h_lora_sanity.py::TestNemotronHLoRA::test_lora_pp2_sanity SKIP (https://nvbugs/6428124) -unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_fp8_lora SKIP (https://nvbugs/6668777) unittest/_torch/moe/test_kimi_k3_situ_moe.py::test_kimi_k3_trtllm_accepts_nvfp4_routed_experts SKIP (https://nvbugs/6765038) unittest/_torch/moe/test_moe_backend.py::test_moe_backend[act=Relu2-e60_k4_h2048_i1408-seq=8-dtype=torch.bfloat16-backend=TRTLLM-quant=NVFP4-routing=Renormalize] SKIP (https://nvbugs/5989912) unittest/_torch/multi_gpu/test_linear.py::test_row_linear_norm_fusion[2-hidden:16-seqlen:2] SKIP (https://nvbugs/6501404) diff --git a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py index cc3ecb3e4e81..7ebacde22297 100644 --- a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py +++ b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py @@ -31,6 +31,7 @@ from tensorrt_llm._torch.peft.lora.config import LoraConfig from tensorrt_llm._torch.peft.lora.layer import LoraLayer, LoraModuleType from tensorrt_llm.executor.request import LoRARequest +from tensorrt_llm.llmapi import KvCacheConfig # HF module name -> block path relative to layers.{idx}. # Attention targets work on all architectures. MLP targets only apply to @@ -110,12 +111,15 @@ def _create_lora_adapter( return output_dir -def _run_with_and_without_lora(model_path, lora_config, lora_dir, prompts): +def _run_with_and_without_lora( + model_path, lora_config, lora_dir, prompts, kv_cache_config: KvCacheConfig | None = None +): """Run inference with and without LoRA, return (lora_outputs, base_outputs).""" with LLM( model=model_path, backend="pytorch", lora_config=lora_config, + kv_cache_config=kv_cache_config or KvCacheConfig(), tensor_parallel_size=1, max_batch_size=4, max_num_tokens=256, @@ -177,6 +181,7 @@ def _run_lora_test( dtype=torch.bfloat16, overlap=False, specialize_cuda_graph=False, + kv_cache_config: KvCacheConfig | None = None, ): """End-to-end helper: create adapter, run inference, assert output differs.""" with tempfile.TemporaryDirectory() as tmpdir: @@ -200,11 +205,14 @@ def _run_lora_test( lora_config, lora_dir, ["The capital of France is", "Hello, how are you"], + kv_cache_config=kv_cache_config, ) _assert_lora_changes_output(out_lora, out_base) -def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules): +def _run_mixed_lora_cuda_graph_test( + model_path, target_modules, trtllm_modules, kv_cache_config: KvCacheConfig +): """Verify base rows remain unchanged when sharing a LoRA-specialized graph.""" with tempfile.TemporaryDirectory() as tmpdir: lora_dir = _create_lora_adapter( @@ -223,6 +231,7 @@ def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules): model=model_path, backend="pytorch", lora_config=lora_config, + kv_cache_config=kv_cache_config, tensor_parallel_size=1, max_batch_size=4, max_num_tokens=256, @@ -235,13 +244,27 @@ def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules): "Hello, how are you", ] lora_request = LoRARequest("test-lora", 0, lora_dir) + mixed_lora_requests = [lora_request, None, lora_request, None] + if kv_cache_config.enable_block_reuse: + # Warm both adapter and base prefixes before either measured call. + # Comparing a cold prefill with a partial cache hit exercises + # different attention shapes, which can change BF16 logprobs. + llm.generate(prompts, sampling, lora_request=mixed_lora_requests) mixed_outputs = llm.generate( prompts, sampling, - lora_request=[lora_request, None, lora_request, None], + lora_request=mixed_lora_requests, ) base_outputs = llm.generate(prompts, sampling) + for mixed_output, base_output in zip(mixed_outputs, base_outputs, strict=True): + expected_cached_tokens = 0 + if kv_cache_config.enable_block_reuse: + # The final prompt token must still be computed to produce logits. + expected_cached_tokens = len(mixed_output.prompt_token_ids) - 1 + assert expected_cached_tokens > 0 + assert mixed_output.cached_tokens == expected_cached_tokens + assert base_output.cached_tokens == expected_cached_tokens for index in (1, 3): _assert_outputs_match(mixed_outputs[index], base_outputs[index]) _assert_lora_changes_output( @@ -286,6 +309,7 @@ class TestQwen3LoRA: @pytest.fixture(autouse=True) def setup(self): self.model_path = f"{llm_models_root()}/Qwen3/Qwen3-0.6B" + self.kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2=True) if not os.path.exists(self.model_path): pytest.skip(f"Model not found: {self.model_path}") @@ -294,6 +318,7 @@ def test_qwen3_bf16_lora(self): self.model_path, {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_fp8_lora(self): @@ -302,6 +327,7 @@ def test_qwen3_fp8_lora(self): {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, dtype=torch.float8_e4m3fn, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_bf16_lora_overlap(self): @@ -310,6 +336,7 @@ def test_qwen3_bf16_lora_overlap(self): {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, overlap=True, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_fp8_lora_overlap(self): @@ -319,6 +346,7 @@ def test_qwen3_fp8_lora_overlap(self): _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, dtype=torch.float8_e4m3fn, overlap=True, + kv_cache_config=self.kv_cache_config, ) def test_qwen3_bf16_lora_cuda_graph_specialization(self): @@ -327,13 +355,17 @@ def test_qwen3_bf16_lora_cuda_graph_specialization(self): {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, specialize_cuda_graph=True, + kv_cache_config=self.kv_cache_config, ) - def test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch(self): + @pytest.mark.parametrize("enable_block_reuse", [False, True], ids=["no_reuse", "warmed_reuse"]) + def test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch(self, enable_block_reuse): + self.kv_cache_config.enable_block_reuse = enable_block_reuse _run_mixed_lora_cuda_graph_test( self.model_path, {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, + kv_cache_config=self.kv_cache_config, ) From a8e3719f4f8205eb09b91a8ad8680a0e53f9898c Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 03:17:26 -0700 Subject: [PATCH 2/7] [None][test] Preserve the existing GB300 Qwen3 FP8 waiver Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- tests/integration/test_lists/waives.txt | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index caff39ff9dc9..982b8dbad0b6 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -170,6 +170,7 @@ full:GB300/accuracy/test_llm_api_pytorch_multimodal.py::TestExaone4_5_33B::test_ full:GB300/disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_attention_dp_overlap[DeepSeek-V3-Lite-fp8] SKIP (https://nvbugs/6581064) full:GB300/disaggregated/test_disaggregated.py::test_disaggregated_logprobs_serving[llama-3.1-8b-instruct] SKIP (https://nvbugs/6275959) full:GB300/unittest/_torch/modeling/test_modeling_gpt_oss.py::test_gpt_oss_trtllmgen[CUTLASS] SKIP (https://nvbugs/6633932) +full:GB300/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_fp8_lora SKIP (https://nvbugs/6668777) full:H100/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=False] SKIP (https://nvbugs/6313072) full:H100/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=True] SKIP (https://nvbugs/6313072) full:H100/accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_dummy_load_format SKIP (https://nvbugs/6626640) From 4ff098b620d95615d44fbfbef06b1911b9f09f92 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 05:50:30 -0700 Subject: [PATCH 3/7] [None][fix] Preserve cold-to-warm Qwen3 LoRA V2 coverage Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../tests_lora_modules/test_qwen3_sanity.py | 42 +++++++++---------- 1 file changed, 21 insertions(+), 21 deletions(-) diff --git a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py index 7ebacde22297..554c77a91ba6 100644 --- a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py +++ b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py @@ -157,8 +157,8 @@ def _assert_lora_changes_output(out_lora, out_base): assert any_differ, "LoRA outputs identical to base model (same tokens AND same logprobs)" -def _assert_outputs_match(actual, expected): - """Assert that two request outputs have identical tokens and logprobs.""" +def _assert_outputs_match(actual, expected, *, logprob_atol=1e-6): + """Require identical tokens and logprobs within the supplied absolute bound.""" actual_completion = actual.outputs[0] expected_completion = expected.outputs[0] assert actual_completion.token_ids == expected_completion.token_ids @@ -170,7 +170,7 @@ def _assert_outputs_match(actual, expected): assert actual_step.keys() == expected_step.keys() for token_id in actual_step: assert actual_step[token_id].logprob == pytest.approx( - expected_step[token_id].logprob, abs=1e-6 + expected_step[token_id].logprob, rel=0, abs=logprob_atol ) @@ -245,32 +245,34 @@ def _run_mixed_lora_cuda_graph_test( ] lora_request = LoRARequest("test-lora", 0, lora_dir) mixed_lora_requests = [lora_request, None, lora_request, None] - if kv_cache_config.enable_block_reuse: - # Warm both adapter and base prefixes before either measured call. - # Comparing a cold prefill with a partial cache hit exercises - # different attention shapes, which can change BF16 logprobs. - llm.generate(prompts, sampling, lora_request=mixed_lora_requests) mixed_outputs = llm.generate( prompts, sampling, lora_request=mixed_lora_requests, ) base_outputs = llm.generate(prompts, sampling) + reused_mixed_outputs = llm.generate(prompts, sampling, lora_request=mixed_lora_requests) - for mixed_output, base_output in zip(mixed_outputs, base_outputs, strict=True): - expected_cached_tokens = 0 - if kv_cache_config.enable_block_reuse: - # The final prompt token must still be computed to produce logits. - expected_cached_tokens = len(mixed_output.prompt_token_ids) - 1 - assert expected_cached_tokens > 0 - assert mixed_output.cached_tokens == expected_cached_tokens - assert base_output.cached_tokens == expected_cached_tokens + # Keep the cold-to-reused transition that exercises final-context + # promotion; a warmup or disabled reuse must not hide a regression. + assert all(output.cached_tokens == 0 for output in mixed_outputs) + for output in base_outputs + reused_mixed_outputs: + assert output.cached_tokens == len(output.prompt_token_ids) - 1 for index in (1, 3): - _assert_outputs_match(mixed_outputs[index], base_outputs[index]) + # Cold prefill and reused final-context decode use different BF16 + # reductions, in both KVCM V1 and V2. Bound the resulting logprob + # drift (exp(0.1) is about 1.105), while keeping tokens identical. + _assert_outputs_match(mixed_outputs[index], base_outputs[index], logprob_atol=0.1) + # With the same cache state, retain the strict LoRA-isolation check. + _assert_outputs_match(reused_mixed_outputs[index], base_outputs[index]) _assert_lora_changes_output( [mixed_outputs[index] for index in (0, 2)], [base_outputs[index] for index in (0, 2)], ) + _assert_lora_changes_output( + [reused_mixed_outputs[index] for index in (0, 2)], + [base_outputs[index] for index in (0, 2)], + ) @pytest.mark.skipif(not torch.cuda.is_available(), reason="LoRA overlap requires CUDA streams.") @@ -309,7 +311,7 @@ class TestQwen3LoRA: @pytest.fixture(autouse=True) def setup(self): self.model_path = f"{llm_models_root()}/Qwen3/Qwen3-0.6B" - self.kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2=True) + self.kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2=True, enable_block_reuse=True) if not os.path.exists(self.model_path): pytest.skip(f"Model not found: {self.model_path}") @@ -358,9 +360,7 @@ def test_qwen3_bf16_lora_cuda_graph_specialization(self): kv_cache_config=self.kv_cache_config, ) - @pytest.mark.parametrize("enable_block_reuse", [False, True], ids=["no_reuse", "warmed_reuse"]) - def test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch(self, enable_block_reuse): - self.kv_cache_config.enable_block_reuse = enable_block_reuse + def test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch(self): _run_mixed_lora_cuda_graph_test( self.model_path, {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, From 259ea8837471955e6ca47544e8b23a1d518d59bd Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 06:33:33 -0700 Subject: [PATCH 4/7] [None][fix] Keep Qwen3 LoRA validation to the original two calls Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../tests_lora_modules/test_qwen3_sanity.py | 20 ++++++------------- 1 file changed, 6 insertions(+), 14 deletions(-) diff --git a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py index 554c77a91ba6..10ec696c63a3 100644 --- a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py +++ b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py @@ -157,8 +157,8 @@ def _assert_lora_changes_output(out_lora, out_base): assert any_differ, "LoRA outputs identical to base model (same tokens AND same logprobs)" -def _assert_outputs_match(actual, expected, *, logprob_atol=1e-6): - """Require identical tokens and logprobs within the supplied absolute bound.""" +def _assert_outputs_match(actual, expected): + """Require identical tokens and logprobs within the cold/warm BF16 bound.""" actual_completion = actual.outputs[0] expected_completion = expected.outputs[0] assert actual_completion.token_ids == expected_completion.token_ids @@ -170,7 +170,7 @@ def _assert_outputs_match(actual, expected, *, logprob_atol=1e-6): assert actual_step.keys() == expected_step.keys() for token_id in actual_step: assert actual_step[token_id].logprob == pytest.approx( - expected_step[token_id].logprob, rel=0, abs=logprob_atol + expected_step[token_id].logprob, rel=0, abs=0.1 ) @@ -244,35 +244,27 @@ def _run_mixed_lora_cuda_graph_test( "Hello, how are you", ] lora_request = LoRARequest("test-lora", 0, lora_dir) - mixed_lora_requests = [lora_request, None, lora_request, None] mixed_outputs = llm.generate( prompts, sampling, - lora_request=mixed_lora_requests, + lora_request=[lora_request, None, lora_request, None], ) base_outputs = llm.generate(prompts, sampling) - reused_mixed_outputs = llm.generate(prompts, sampling, lora_request=mixed_lora_requests) # Keep the cold-to-reused transition that exercises final-context # promotion; a warmup or disabled reuse must not hide a regression. assert all(output.cached_tokens == 0 for output in mixed_outputs) - for output in base_outputs + reused_mixed_outputs: + for output in base_outputs: assert output.cached_tokens == len(output.prompt_token_ids) - 1 for index in (1, 3): # Cold prefill and reused final-context decode use different BF16 # reductions, in both KVCM V1 and V2. Bound the resulting logprob # drift (exp(0.1) is about 1.105), while keeping tokens identical. - _assert_outputs_match(mixed_outputs[index], base_outputs[index], logprob_atol=0.1) - # With the same cache state, retain the strict LoRA-isolation check. - _assert_outputs_match(reused_mixed_outputs[index], base_outputs[index]) + _assert_outputs_match(mixed_outputs[index], base_outputs[index]) _assert_lora_changes_output( [mixed_outputs[index] for index in (0, 2)], [base_outputs[index] for index in (0, 2)], ) - _assert_lora_changes_output( - [reused_mixed_outputs[index] for index in (0, 2)], - [base_outputs[index] for index in (0, 2)], - ) @pytest.mark.skipif(not torch.cuda.is_available(), reason="LoRA overlap requires CUDA streams.") From c70c70fcdc64852156bca158d1769ea3ef88ace4 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:08:27 -0700 Subject: [PATCH 5/7] [None][fix] Keep strict Qwen3 LoRA checks without partial reuse Disable partial KV cache reuse only for the mixed LoRA CUDA graph case so both original calls use full prefill. Restore the original exact-token and logprob assertions. Validated all six dense LoRA cases on both H100 and B200. Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../tests_lora_modules/test_qwen3_sanity.py | 16 ++++++---------- 1 file changed, 6 insertions(+), 10 deletions(-) diff --git a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py index 10ec696c63a3..2a6631401e96 100644 --- a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py +++ b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py @@ -158,7 +158,7 @@ def _assert_lora_changes_output(out_lora, out_base): def _assert_outputs_match(actual, expected): - """Require identical tokens and logprobs within the cold/warm BF16 bound.""" + """Assert that two request outputs have identical tokens and logprobs.""" actual_completion = actual.outputs[0] expected_completion = expected.outputs[0] assert actual_completion.token_ids == expected_completion.token_ids @@ -170,7 +170,7 @@ def _assert_outputs_match(actual, expected): assert actual_step.keys() == expected_step.keys() for token_id in actual_step: assert actual_step[token_id].logprob == pytest.approx( - expected_step[token_id].logprob, rel=0, abs=0.1 + expected_step[token_id].logprob, abs=1e-6 ) @@ -251,15 +251,9 @@ def _run_mixed_lora_cuda_graph_test( ) base_outputs = llm.generate(prompts, sampling) - # Keep the cold-to-reused transition that exercises final-context - # promotion; a warmup or disabled reuse must not hide a regression. assert all(output.cached_tokens == 0 for output in mixed_outputs) - for output in base_outputs: - assert output.cached_tokens == len(output.prompt_token_ids) - 1 + assert all(output.cached_tokens == 0 for output in base_outputs) for index in (1, 3): - # Cold prefill and reused final-context decode use different BF16 - # reductions, in both KVCM V1 and V2. Bound the resulting logprob - # drift (exp(0.1) is about 1.105), while keeping tokens identical. _assert_outputs_match(mixed_outputs[index], base_outputs[index]) _assert_lora_changes_output( [mixed_outputs[index] for index in (0, 2)], @@ -353,11 +347,13 @@ def test_qwen3_bf16_lora_cuda_graph_specialization(self): ) def test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch(self): + # Keep these short prompts on the same context attention path so the + # strict comparison measures LoRA isolation without cold/warm drift. _run_mixed_lora_cuda_graph_test( self.model_path, {**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES}, _ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES, - kv_cache_config=self.kv_cache_config, + kv_cache_config=self.kv_cache_config.model_copy(update={"enable_partial_reuse": False}), ) From eb25721c37c81e0be67f68b2c68e6d2d12e78334 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:12:21 -0700 Subject: [PATCH 6/7] [None][test] Explain full-prefill requirement for strict LoRA checks Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../_torch/modules/tests_lora_modules/test_qwen3_sanity.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py index 2a6631401e96..f8e6522d11c2 100644 --- a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py +++ b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py @@ -251,6 +251,9 @@ def _run_mixed_lora_cuda_graph_test( ) base_outputs = llm.generate(prompts, sampling) + # Keep both calls on full prefill: partial reuse can select a different + # attention kernel, whose numerical drift can fail this strict token and + # logprob comparison even when LoRA correctly leaves base rows unchanged. assert all(output.cached_tokens == 0 for output in mixed_outputs) assert all(output.cached_tokens == 0 for output in base_outputs) for index in (1, 3): From 58e40cbfddc00e7b06b6f38e02fd2ff8fa3d3818 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:52:45 -0700 Subject: [PATCH 7/7] [None][test] Keep mixed Qwen3 LoRA decode batches aligned Ignore EOS in the mixed CUDA graph isolation test so early adapter completion cannot change batch shape and BF16 GEMM rounding relative to the all-base reference. Preserve the existing strict assertions. Validated all six TestQwen3LoRA cases on H100, B200, and B300 using CI 60618 artifacts. Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../_torch/modules/tests_lora_modules/test_qwen3_sanity.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py index f8e6522d11c2..4f5b2d95f416 100644 --- a/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py +++ b/tests/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py @@ -236,7 +236,9 @@ def _run_mixed_lora_cuda_graph_test( max_batch_size=4, max_num_tokens=256, ) as llm: - sampling = SamplingParams(max_tokens=20, temperature=0.0, logprobs=0) + # Prevent adapter EOS from shrinking the mixed batch and changing + # BF16 GEMM rounding relative to the all-base reference. + sampling = SamplingParams(max_tokens=20, temperature=0.0, logprobs=0, ignore_eos=True) prompts = [ "The capital of France is", "The capital of France is",