Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion tests/integration/test_lists/waives.txt
Original file line number Diff line number Diff line change
Expand Up @@ -170,6 +170,7 @@ full:GB300/accuracy/test_llm_api_pytorch_multimodal.py::TestExaone4_5_33B::test_
full:GB300/disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_attention_dp_overlap[DeepSeek-V3-Lite-fp8] SKIP (https://nvbugs/6581064)
full:GB300/disaggregated/test_disaggregated.py::test_disaggregated_logprobs_serving[llama-3.1-8b-instruct] SKIP (https://nvbugs/6275959)
full:GB300/unittest/_torch/modeling/test_modeling_gpt_oss.py::test_gpt_oss_trtllmgen[CUTLASS] SKIP (https://nvbugs/6633932)
full:GB300/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_fp8_lora SKIP (https://nvbugs/6668777)
full:H100/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=False] SKIP (https://nvbugs/6313072)
full:H100/accuracy/test_disaggregated_serving.py::TestDeepSeekV3Lite::test_auto_dtype[mtp_nextn=2-overlap_scheduler=True] SKIP (https://nvbugs/6313072)
full:H100/accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_dummy_load_format SKIP (https://nvbugs/6626640)
Expand Down Expand Up @@ -266,7 +267,6 @@ unittest/_torch/modeling/test_modeling_nemotron_h.py::test_nemotron_h_breakable_
unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py::test_nemotron_nano_v2_vl_image_batch_equivalence SKIP (https://nvbugs/6758946)
unittest/_torch/modeling/test_modeling_nemotron_nano_v2_vl.py::test_nemotron_nano_v2_vl_video_batch_equivalence SKIP (https://nvbugs/6625695)
unittest/_torch/modules/tests_lora_modules/test_nemotron_h_lora_sanity.py::TestNemotronHLoRA::test_lora_pp2_sanity SKIP (https://nvbugs/6428124)
unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_fp8_lora SKIP (https://nvbugs/6668777)
unittest/_torch/moe/test_kimi_k3_situ_moe.py::test_kimi_k3_trtllm_accepts_nvfp4_routed_experts SKIP (https://nvbugs/6765038)
unittest/_torch/moe/test_moe_backend.py::test_moe_backend[act=Relu2-e60_k4_h2048_i1408-seq=8-dtype=torch.bfloat16-backend=TRTLLM-quant=NVFP4-routing=Renormalize] SKIP (https://nvbugs/5989912)
unittest/_torch/multi_gpu/test_linear.py::test_row_linear_norm_fusion[2-hidden:16-seqlen:2] SKIP (https://nvbugs/6501404)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@
from tensorrt_llm._torch.peft.lora.config import LoraConfig
from tensorrt_llm._torch.peft.lora.layer import LoraLayer, LoraModuleType
from tensorrt_llm.executor.request import LoRARequest
from tensorrt_llm.llmapi import KvCacheConfig

# HF module name -> block path relative to layers.{idx}.
# Attention targets work on all architectures. MLP targets only apply to
Expand Down Expand Up @@ -110,12 +111,15 @@ def _create_lora_adapter(
return output_dir


def _run_with_and_without_lora(model_path, lora_config, lora_dir, prompts):
def _run_with_and_without_lora(
model_path, lora_config, lora_dir, prompts, kv_cache_config: KvCacheConfig | None = None
):
"""Run inference with and without LoRA, return (lora_outputs, base_outputs)."""
with LLM(
model=model_path,
backend="pytorch",
lora_config=lora_config,
kv_cache_config=kv_cache_config or KvCacheConfig(),
tensor_parallel_size=1,
max_batch_size=4,
max_num_tokens=256,
Expand Down Expand Up @@ -177,6 +181,7 @@ def _run_lora_test(
dtype=torch.bfloat16,
overlap=False,
specialize_cuda_graph=False,
kv_cache_config: KvCacheConfig | None = None,
):
"""End-to-end helper: create adapter, run inference, assert output differs."""
with tempfile.TemporaryDirectory() as tmpdir:
Expand All @@ -200,11 +205,14 @@ def _run_lora_test(
lora_config,
lora_dir,
["The capital of France is", "Hello, how are you"],
kv_cache_config=kv_cache_config,
)
_assert_lora_changes_output(out_lora, out_base)


def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules):
def _run_mixed_lora_cuda_graph_test(
model_path, target_modules, trtllm_modules, kv_cache_config: KvCacheConfig
):
"""Verify base rows remain unchanged when sharing a LoRA-specialized graph."""
with tempfile.TemporaryDirectory() as tmpdir:
lora_dir = _create_lora_adapter(
Expand All @@ -223,11 +231,14 @@ def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules):
model=model_path,
backend="pytorch",
lora_config=lora_config,
kv_cache_config=kv_cache_config,
tensor_parallel_size=1,
max_batch_size=4,
max_num_tokens=256,
) as llm:
sampling = SamplingParams(max_tokens=20, temperature=0.0, logprobs=0)
# Prevent adapter EOS from shrinking the mixed batch and changing
# BF16 GEMM rounding relative to the all-base reference.
sampling = SamplingParams(max_tokens=20, temperature=0.0, logprobs=0, ignore_eos=True)
prompts = [
"The capital of France is",
"The capital of France is",
Expand All @@ -242,6 +253,11 @@ def _run_mixed_lora_cuda_graph_test(model_path, target_modules, trtllm_modules):
)
base_outputs = llm.generate(prompts, sampling)

# Keep both calls on full prefill: partial reuse can select a different
# attention kernel, whose numerical drift can fail this strict token and
# logprob comparison even when LoRA correctly leaves base rows unchanged.
assert all(output.cached_tokens == 0 for output in mixed_outputs)
assert all(output.cached_tokens == 0 for output in base_outputs)
for index in (1, 3):
_assert_outputs_match(mixed_outputs[index], base_outputs[index])
_assert_lora_changes_output(
Expand Down Expand Up @@ -286,6 +302,7 @@ class TestQwen3LoRA:
@pytest.fixture(autouse=True)
def setup(self):
self.model_path = f"{llm_models_root()}/Qwen3/Qwen3-0.6B"
self.kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2=True, enable_block_reuse=True)
if not os.path.exists(self.model_path):
pytest.skip(f"Model not found: {self.model_path}")

Expand All @@ -294,6 +311,7 @@ def test_qwen3_bf16_lora(self):
self.model_path,
{**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES},
_ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES,
kv_cache_config=self.kv_cache_config,
)

def test_qwen3_fp8_lora(self):
Expand All @@ -302,6 +320,7 @@ def test_qwen3_fp8_lora(self):
{**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES},
_ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES,
dtype=torch.float8_e4m3fn,
kv_cache_config=self.kv_cache_config,
)

def test_qwen3_bf16_lora_overlap(self):
Expand All @@ -310,6 +329,7 @@ def test_qwen3_bf16_lora_overlap(self):
{**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES},
_ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES,
overlap=True,
kv_cache_config=self.kv_cache_config,
)

def test_qwen3_fp8_lora_overlap(self):
Expand All @@ -319,6 +339,7 @@ def test_qwen3_fp8_lora_overlap(self):
_ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES,
dtype=torch.float8_e4m3fn,
overlap=True,
kv_cache_config=self.kv_cache_config,
)

def test_qwen3_bf16_lora_cuda_graph_specialization(self):
Expand All @@ -327,13 +348,17 @@ def test_qwen3_bf16_lora_cuda_graph_specialization(self):
{**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES},
_ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES,
specialize_cuda_graph=True,
kv_cache_config=self.kv_cache_config,
)

def test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch(self):
# Keep these short prompts on the same context attention path so the
# strict comparison measures LoRA isolation without cold/warm drift.
_run_mixed_lora_cuda_graph_test(
self.model_path,
{**_ATTN_LORA_MODULES, **_MLP_LORA_MODULES},
_ATTN_TRTLLM_MODULES + _MLP_TRTLLM_MODULES,
kv_cache_config=self.kv_cache_config.model_copy(update={"enable_partial_reuse": False}),
)


Expand Down
Loading