From 42daa875a0463e37ed5e498ad864b904fa900629 Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Mon, 17 Aug 2026 08:12:14 -0700
Subject: [PATCH 01/16] [None][feat] Default to KVCM V2 and Python transceiver
Enable KV cache manager V2 and the Python cache transceiver by default, and migrate supported examples and tests to the V2 path.
Keep explicit V1 and C++ settings only for legacy transport, compatibility, and dedicated coverage.
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
.../service_discovery_example/launch.slurm | 6 ++--
.../ctx_extra-llm-api-config.yaml | 3 +-
.../gen_extra-llm-api-config.yaml | 3 +-
examples/dwdp/reproduce.py | 6 ++--
examples/llm-api/quickstart_advanced.py | 4 +--
.../disaggregated/disagg_serving_local.sh | 4 +--
.../slurm_scripts/kimi-k2-thinking.yaml | 8 ++---
tensorrt_llm/_torch/pyexecutor/_util.py | 3 ++
tensorrt_llm/llmapi/llm_args.py | 20 +++++------
.../accuracy/test_disaggregated_serving.py | 15 +++++---
.../test_dwdp_disaggregated_serving.py | 21 +++++-------
...ig_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml | 19 +++++++++++
...tp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml | 2 ++
...ig_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml | 4 +++
...isagg_config_llama4_kv_cache_overflow.yaml | 4 +++
.../test_configs/disagg_config_metrics.yaml | 8 +++--
.../disaggregated/test_disaggregated_etcd.py | 4 +--
.../test_disaggregated_single_gpu.py | 17 +++++++---
...pseek_v32_fp4_2_nodes_grace_blackwell.yaml | 1 +
.../dynamo_k25_thinking_fp4_blackwell.yaml | 4 +--
.../precheck_config.py | 2 +-
..._ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml | 2 ++
...tx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml | 2 ++
...1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml | 2 ++
.../test_disagg_inflight_cancel_gate.py | 8 ++---
.../kv_cache/test_kv_cache_estimation.py | 30 ++++++++++++++++
.../kv_cache/test_mamba_cache_manager.py | 14 ++++++--
.../_torch/modeling/test_modeling_gpt_oss.py | 5 +--
.../_torch/speculative/test_eagle3.py | 9 +++--
tests/unittest/llmapi/test_async_llm.py | 2 +-
tests/unittest/llmapi/test_llm_args.py | 34 +++++++++++--------
tests/unittest/llmapi/test_llm_pytorch.py | 29 +++++++++++-----
.../llmapi/test_quickstart_advanced.py | 4 +--
.../test_cache_transceiver_precheck_config.py | 6 ++--
.../others/test_kv_cache_transceiver.py | 9 ++++-
35 files changed, 214 insertions(+), 100 deletions(-)
diff --git a/examples/disaggregated/slurm/service_discovery_example/launch.slurm b/examples/disaggregated/slurm/service_discovery_example/launch.slurm
index 60bac45b2e2d..ed584a50533d 100644
--- a/examples/disaggregated/slurm/service_discovery_example/launch.slurm
+++ b/examples/disaggregated/slurm/service_discovery_example/launch.slurm
@@ -44,16 +44,14 @@ cat >${work_path}/ctx_config.yaml << EOL
disable_overlap_scheduler: True
internal_request_auth_key: ${internal_request_auth_key}
cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
+ backend: NIXL
max_tokens_in_buffer: 2048
EOL
cat >${work_path}/gen_config.yaml << EOL
internal_request_auth_key: ${internal_request_auth_key}
cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
+ backend: NIXL
max_tokens_in_buffer: 2048
EOL
diff --git a/examples/disaggregated/slurm/simple_example/ctx_extra-llm-api-config.yaml b/examples/disaggregated/slurm/simple_example/ctx_extra-llm-api-config.yaml
index cce87b3ea2f1..a43d34f2d485 100644
--- a/examples/disaggregated/slurm/simple_example/ctx_extra-llm-api-config.yaml
+++ b/examples/disaggregated/slurm/simple_example/ctx_extra-llm-api-config.yaml
@@ -2,6 +2,5 @@
# not yet supported in disaggregated context server architectures.
disable_overlap_scheduler: True
cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
+ backend: NIXL
max_tokens_in_buffer: 2048
diff --git a/examples/disaggregated/slurm/simple_example/gen_extra-llm-api-config.yaml b/examples/disaggregated/slurm/simple_example/gen_extra-llm-api-config.yaml
index 5324fd439098..61bbe07a7d94 100644
--- a/examples/disaggregated/slurm/simple_example/gen_extra-llm-api-config.yaml
+++ b/examples/disaggregated/slurm/simple_example/gen_extra-llm-api-config.yaml
@@ -1,4 +1,3 @@
cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
+ backend: NIXL
max_tokens_in_buffer: 2048
diff --git a/examples/dwdp/reproduce.py b/examples/dwdp/reproduce.py
index f214329ebf7d..959e4f535805 100644
--- a/examples/dwdp/reproduce.py
+++ b/examples/dwdp/reproduce.py
@@ -220,8 +220,7 @@ def build_worker_config(experiment: Dict[str, Any]) -> Dict[str, Any]:
"backend": "CUTEDSL",
},
"cache_transceiver_config": {
- "backend": "UCX",
- "transceiver_runtime": "CPP",
+ "backend": "NIXL",
"max_tokens_in_buffer": max_tokens_in_buffer,
},
"num_postprocess_workers": 4,
@@ -256,8 +255,7 @@ def build_worker_config(experiment: Dict[str, Any]) -> Dict[str, Any]:
"free_gpu_memory_fraction": 0.3,
},
"cache_transceiver_config": {
- "backend": "UCX",
- "transceiver_runtime": "CPP",
+ "backend": "NIXL",
"max_tokens_in_buffer": max_tokens_in_buffer,
},
"moe_config": {
diff --git a/examples/llm-api/quickstart_advanced.py b/examples/llm-api/quickstart_advanced.py
index a167ec0ae3e7..371deac1afa1 100644
--- a/examples/llm-api/quickstart_advanced.py
+++ b/examples/llm-api/quickstart_advanced.py
@@ -151,11 +151,11 @@ def add_llm_args(parser):
action='store_true')
parser.add_argument(
'--use_kv_cache_manager_v2',
- default='auto',
+ default=True,
type=_parse_kv_cache_manager_v2,
metavar='{auto,true,false}',
help=
- 'Whether to use KVCacheManagerV2 for KV cache management (PyTorch backend). Defaults to model-specific auto selection.',
+ 'Whether to use KVCacheManagerV2 for KV cache management (PyTorch backend). Defaults to true; use auto for model-specific selection.',
)
# Runtime
diff --git a/examples/ray_orchestrator/disaggregated/disagg_serving_local.sh b/examples/ray_orchestrator/disaggregated/disagg_serving_local.sh
index 40ee6745e693..a49e2f404b4a 100644
--- a/examples/ray_orchestrator/disaggregated/disagg_serving_local.sh
+++ b/examples/ray_orchestrator/disaggregated/disagg_serving_local.sh
@@ -8,7 +8,7 @@ ATTACH_MODE=false
MODEL_DIR="TinyLlama/TinyLlama-1.1B-Chat-v1.0"
TP_SIZE=1
TRANSCEIVER_BACKEND="NIXL"
-TRANSCEIVER_RUNTIME="CPP"
+TRANSCEIVER_RUNTIME="PYTHON"
USAGE="Usage: $0 [--executor ray|mpi] [--attach] [--model model_dir] [--tp_size N] [--transceiver_backend UCX|NIXL] [--transceiver_runtime CPP|PYTHON] [--help]"
while [[ $# -gt 0 ]]; do
@@ -45,7 +45,7 @@ while [[ $# -gt 0 ]]; do
echo " --model model_dir Model directory (default: TinyLlama/TinyLlama-1.1B-Chat-v1.0)"
echo " --tp_size N Tensor parallel size (default: 1)"
echo " --transceiver_backend UCX|NIXL Cache-transceiver backend (default: NIXL)"
- echo " --transceiver_runtime CPP|PYTHON Cache transceiver runtime (default: CPP)"
+ echo " --transceiver_runtime CPP|PYTHON Cache transceiver runtime (default: PYTHON)"
echo " --help, -h Show this help message"
exit 0
;;
diff --git a/examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml b/examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml
index 17e0f96a381f..789e3ad847ed 100644
--- a/examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml
+++ b/examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml
@@ -72,8 +72,8 @@ worker_config:
num_slots: 416
layer_updates_per_iter: 1
cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
+ backend: NIXL
+ transceiver_runtime: PYTHON
max_tokens_in_buffer: 8448
stream_interval: 20
num_postprocess_workers: 4
@@ -94,7 +94,7 @@ worker_config:
free_gpu_memory_fraction: 0.75
dtype: fp8
cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
+ backend: NIXL
+ transceiver_runtime: PYTHON
max_tokens_in_buffer: 8448
trust_remote_code: true
diff --git a/tensorrt_llm/_torch/pyexecutor/_util.py b/tensorrt_llm/_torch/pyexecutor/_util.py
index 660ffe11b601..73759bfb55ab 100644
--- a/tensorrt_llm/_torch/pyexecutor/_util.py
+++ b/tensorrt_llm/_torch/pyexecutor/_util.py
@@ -1399,6 +1399,9 @@ def try_prepare_estimation(self) -> bool:
self._skip_est = True
model_config = self._model_engine.model.model_config
if model_config.attn_backend == "VANILLA":
+ if (self._is_kv_cache_manager_v2
+ and 'cp_type' not in self._mapping.cp_config):
+ self._skip_est = True
estimating_kv_cache = False
logger.info(
"KV cache size estimation is not supported for Vanilla attention backend, disable it."
diff --git a/tensorrt_llm/llmapi/llm_args.py b/tensorrt_llm/llmapi/llm_args.py
index d596c1d38a4e..e3aa05cd3a9f 100644
--- a/tensorrt_llm/llmapi/llm_args.py
+++ b/tensorrt_llm/llmapi/llm_args.py
@@ -4305,7 +4305,7 @@ class KvCacheConfig(StrictBaseModel, PybindMirror):
description="Configuration for reusable Mamba state snapshots.")
use_kv_cache_manager_v2: bool | Literal["auto"] = Field(
- default="auto",
+ default=True,
status="prototype",
description=
"Whether to use the KV cache manager v2 (experimental). 'auto' uses "
@@ -4624,17 +4624,17 @@ class CacheTransceiverConfig(StrictBaseModel, PybindMirror):
"The communication backend type to use for the cache transceiver.")
transceiver_runtime: Optional[Literal["CPP", "PYTHON", "auto"]] = Field(
- default="auto",
- description=
- "The runtime implementation. 'auto' (default) adopts the model's "
- "preferred runtime when it declares one; otherwise it selects the "
- "Python transceiver, falling back to the C++ transceiver only when "
- "this config itself rules it out (non-NIXL backend or a null "
- "kv_transfer_timeout_ms) — any other incompatibility fails at "
+ default="PYTHON",
+ description=
+ "The runtime implementation. 'PYTHON' (default) selects the Python "
+ "transceiver, while 'CPP' selects the C++ transceiver. 'auto' adopts "
+ "the model's preferred runtime when it declares one; otherwise it "
+ "selects the Python transceiver, falling back to the C++ transceiver "
+ "only when this config itself rules it out (non-NIXL backend or a "
+ "null kv_transfer_timeout_ms) — any other incompatibility fails at "
"transceiver creation. The fallback is decided independently on "
"each server and is only logged, not surfaced, so keep context and "
- "generation server configurations consistent. 'CPP' selects the C++ "
- "transceiver, 'PYTHON' the Python transceiver. None is equivalent "
+ "generation server configurations consistent. None is equivalent "
"to 'CPP'. 'auto' is resolved on the PyTorch backend's standard "
"model-loading path.")
diff --git a/tests/integration/defs/accuracy/test_disaggregated_serving.py b/tests/integration/defs/accuracy/test_disaggregated_serving.py
index d6f8287c998e..e82d48d1fafa 100644
--- a/tests/integration/defs/accuracy/test_disaggregated_serving.py
+++ b/tests/integration/defs/accuracy/test_disaggregated_serving.py
@@ -698,7 +698,7 @@ def run_parallel_test(model_name: str,
test_sets: List[LlmapiAccuracyTestHarness],
ctx_model: str = None,
gen_model: str = None,
- cache_transceiver_backend: str = "DEFAULT",
+ cache_transceiver_backend: str = "NIXL",
trust_remote_code: bool = False,
quant_algo: str = None,
kv_cache_quant_algo: str = None,
@@ -896,6 +896,7 @@ def test_auto_dtype_with_helix(self, comms_medium, cuda_graph_config,
"enable_block_reuse": False,
"enable_partial_reuse": False,
"tokens_per_block": 32,
+ "use_kv_cache_manager_v2": False,
}
ctx_server_config = {
"pipeline_parallel_size": 1,
@@ -913,6 +914,7 @@ def test_auto_dtype_with_helix(self, comms_medium, cuda_graph_config,
# adopted verbatim and fail at creation on a non-NIXL backend.
"cache_transceiver_config": {
"backend": "DEFAULT",
+ "transceiver_runtime": "CPP",
"max_tokens_in_buffer": 8192,
"transceiver_runtime": "CPP",
},
@@ -934,6 +936,7 @@ def test_auto_dtype_with_helix(self, comms_medium, cuda_graph_config,
"cuda_graph_config": cuda_graph_config,
"cache_transceiver_config": {
"backend": "DEFAULT",
+ "transceiver_runtime": "CPP",
"max_tokens_in_buffer": 8192,
"transceiver_runtime": "CPP",
},
@@ -1390,7 +1393,7 @@ def test_auto_dtype(self, overlap_scheduler, enable_partial_reuse):
"disable_overlap_scheduler": True,
"cuda_graph_config": None,
"cache_transceiver_config": {
- "backend": "DEFAULT",
+ "backend": "NIXL",
"max_tokens_in_buffer": 4096
},
"kv_cache_config": kv_cache_config,
@@ -1399,7 +1402,7 @@ def test_auto_dtype(self, overlap_scheduler, enable_partial_reuse):
"disable_overlap_scheduler": overlap_scheduler,
"cuda_graph_config": None,
"cache_transceiver_config": {
- "backend": "DEFAULT",
+ "backend": "NIXL",
"max_tokens_in_buffer": 4096
},
"kv_cache_config": kv_cache_config,
@@ -1432,7 +1435,7 @@ def _test_chunked_prefill_helper(self, *, ctx_pp: int):
"disable_overlap_scheduler": True,
"cuda_graph_config": None,
"cache_transceiver_config": {
- "backend": "DEFAULT",
+ "backend": "NIXL",
"max_tokens_in_buffer": 4096
},
"enable_chunked_prefill": True,
@@ -1443,7 +1446,7 @@ def _test_chunked_prefill_helper(self, *, ctx_pp: int):
gen_server_config = {
"cuda_graph_config": None,
"cache_transceiver_config": {
- "backend": "DEFAULT",
+ "backend": "NIXL",
"max_tokens_in_buffer": 4096
},
"max_batch_size": max_batch_size,
@@ -1487,9 +1490,11 @@ def _run_helix_test(self, comms_medium, cuda_graph_config, gen_pp, gen_tp,
"enable_block_reuse": False,
"enable_partial_reuse": False,
"tokens_per_block": 32,
+ "use_kv_cache_manager_v2": False,
}
cache_transceiver_config = {
"backend": "DEFAULT",
+ "transceiver_runtime": "CPP",
"max_tokens_in_buffer": 8192,
}
ctx_server_config = {
diff --git a/tests/integration/defs/accuracy/test_dwdp_disaggregated_serving.py b/tests/integration/defs/accuracy/test_dwdp_disaggregated_serving.py
index 801a6858706d..eab54355d695 100644
--- a/tests/integration/defs/accuracy/test_dwdp_disaggregated_serving.py
+++ b/tests/integration/defs/accuracy/test_dwdp_disaggregated_serving.py
@@ -1,3 +1,6 @@
+# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+
"""DWDP disaggregated serving accuracy tests.
Separated from test_disaggregated_serving.py to isolate MPI-dependent test
@@ -228,8 +231,7 @@ def test_dwdp_accuracy(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
- "backend": "UCX",
- "transceiver_runtime": "CPP",
+ "backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
@@ -261,8 +263,7 @@ def test_dwdp_accuracy(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
- "backend": "UCX",
- "transceiver_runtime": "CPP",
+ "backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
@@ -345,8 +346,7 @@ def test_dwdp_accuracy_contention_opt(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
- "backend": "UCX",
- "transceiver_runtime": "CPP",
+ "backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
@@ -379,8 +379,7 @@ def test_dwdp_accuracy_contention_opt(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
- "backend": "UCX",
- "transceiver_runtime": "CPP",
+ "backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
@@ -482,8 +481,7 @@ def test_dwdp_accuracy_mode_b_overlap(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
- "backend": "UCX",
- "transceiver_runtime": "CPP",
+ "backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
@@ -519,8 +517,7 @@ def test_dwdp_accuracy_mode_b_overlap(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
- "backend": "UCX",
- "transceiver_runtime": "CPP",
+ "backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml
index e43517965b8c..18897a334314 100644
--- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml
+++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml
@@ -1,3 +1,18 @@
+# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
hostname: localhost
model: DeepSeek-V3-Lite/fp8
free_gpu_memory_fraction: 0.1
@@ -8,6 +23,8 @@ context_servers:
num_instances: 1
tensor_parallel_size: 1
pipeline_parallel_size: 1
+ kv_cache_config:
+ use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
@@ -15,6 +32,8 @@ generation_servers:
num_instances: 1
tensor_parallel_size: 1
pipeline_parallel_size: 1
+ kv_cache_config:
+ use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml
index 9fa5a4a5a1b0..87a04bde64f1 100644
--- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml
+++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml
@@ -10,6 +10,7 @@ context_servers:
enable_block_reuse: false
enable_partial_reuse: false
tokens_per_block: 32
+ use_kv_cache_manager_v2: false
tensor_parallel_size: 2
pipeline_parallel_size: 1
cache_transceiver_config:
@@ -28,6 +29,7 @@ generation_servers:
enable_block_reuse: false
enable_partial_reuse: false
tokens_per_block: 32
+ use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml
index 689f7d53f64b..cff98e854555 100644
--- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml
+++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml
@@ -7,6 +7,8 @@ context_servers:
num_instances: 1
tensor_parallel_size: 2
pipeline_parallel_size: 1
+ kv_cache_config:
+ use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: MPI
transceiver_runtime: CPP
@@ -14,6 +16,8 @@ generation_servers:
num_instances: 1
tensor_parallel_size: 2
pipeline_parallel_size: 1
+ kv_cache_config:
+ use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: MPI
transceiver_runtime: CPP
diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml
index fa65a710981c..ee04da6504ed 100644
--- a/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml
+++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml
@@ -16,10 +16,12 @@ context_servers:
kv_cache_config:
enable_block_reuse: false
free_gpu_memory_fraction: 0.3
+ use_kv_cache_manager_v2: false
disable_overlap_scheduler: true
cuda_graph_config: null
cache_transceiver_config:
backend: UCX
+ transceiver_runtime: CPP
# Intentionally small to reproduce buffer overflow bug
max_tokens_in_buffer: 2048
generation_servers:
@@ -37,9 +39,11 @@ generation_servers:
kv_cache_config:
enable_block_reuse: false
free_gpu_memory_fraction: 0.3
+ use_kv_cache_manager_v2: false
disable_overlap_scheduler: true
cuda_graph_config: null
cache_transceiver_config:
backend: UCX
+ transceiver_runtime: CPP
# Intentionally small to reproduce buffer overflow bug
max_tokens_in_buffer: 2048
diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml
index a15f68e6381c..a4f0f896e6a2 100644
--- a/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml
+++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml
@@ -12,15 +12,19 @@ context_servers:
pipeline_parallel_size: 1
return_perf_metrics: True
perf_metrics_max_requests: 1000
+ kv_cache_config:
+ use_kv_cache_manager_v2: auto
cache_transceiver_config:
backend: DEFAULT
- transceiver_runtime: CPP
+ transceiver_runtime: auto
generation_servers:
num_instances: 1
tensor_parallel_size: 1
pipeline_parallel_size: 1
return_perf_metrics: True
perf_metrics_max_requests: 1000
+ kv_cache_config:
+ use_kv_cache_manager_v2: auto
cache_transceiver_config:
backend: DEFAULT
- transceiver_runtime: CPP
+ transceiver_runtime: auto
diff --git a/tests/integration/defs/disaggregated/test_disaggregated_etcd.py b/tests/integration/defs/disaggregated/test_disaggregated_etcd.py
index ec7956ed7393..2e46a2c03a73 100644
--- a/tests/integration/defs/disaggregated/test_disaggregated_etcd.py
+++ b/tests/integration/defs/disaggregated/test_disaggregated_etcd.py
@@ -88,7 +88,7 @@ def start_context_server(config,
server_env = env.copy() if env else os.environ.copy()
server_env["CUDA_VISIBLE_DEVICES"] = str(gpu_id)
- server_env["TRTLLM_USE_UCX_KVCACHE"] = "1"
+ server_env["TRTLLM_USE_NIXL_KVCACHE"] = "1"
server_env["UCX_TLS"] = get_ucx_tls()
logger.info(f"Starting CONTEXT server on GPU {gpu_id} (port {port})...")
@@ -115,7 +115,7 @@ def start_generation_server(config,
server_env = env.copy() if env else os.environ.copy()
server_env["CUDA_VISIBLE_DEVICES"] = str(gpu_id)
- server_env["TRTLLM_USE_UCX_KVCACHE"] = "1"
+ server_env["TRTLLM_USE_NIXL_KVCACHE"] = "1"
server_env["UCX_TLS"] = get_ucx_tls()
logger.info(f"Starting GENERATION server on GPU {gpu_id} (port {port})...")
diff --git a/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py b/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py
index da8aaeefd6ea..d1e9b98db876 100644
--- a/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py
+++ b/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py
@@ -1,3 +1,6 @@
+# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+
import asyncio
import os
import pickle
@@ -596,7 +599,9 @@ def test_disaggregated_spec_dec_batch_slot_limit(model, spec_dec_model_path,
kv_cache_configs = [
KvCacheConfig(max_tokens=128,
enable_block_reuse=False,
- free_gpu_memory_fraction=0.4) for _ in range(2)
+ free_gpu_memory_fraction=0.4,
+ use_kv_cache_manager_v2=eagle3_one_model)
+ for _ in range(2)
]
cache_transceiver_configs = [
CacheTransceiverConfig(backend="DEFAULT") for _ in range(2)
@@ -1050,8 +1055,9 @@ def test_arbitrary_kv_cache_transfer(model, generation_overlap):
cuda_graph_config=CudaGraphConfig()))
kv_cache_configs = [
- KvCacheConfig(max_tokens=2048 * 8, enable_block_reuse=True)
- for _ in range(2)
+ KvCacheConfig(max_tokens=2048 * 8,
+ enable_block_reuse=True,
+ use_kv_cache_manager_v2=False) for _ in range(2)
]
# Arbitrary transfer uses the C++ serialized DataTransceiverState protocol.
cache_transceiver_configs = [
@@ -1210,8 +1216,9 @@ def test_arbitrary_kv_cache_transfer_missing_blocks(model, generation_overlap):
cuda_graph_config=CudaGraphConfig()))
kv_cache_configs = [
- KvCacheConfig(max_tokens=2048 * 8, enable_block_reuse=True)
- for _ in range(2)
+ KvCacheConfig(max_tokens=2048 * 8,
+ enable_block_reuse=True,
+ use_kv_cache_manager_v2=False) for _ in range(2)
]
# Arbitrary transfer uses the C++ serialized DataTransceiverState protocol.
cache_transceiver_configs = [
diff --git a/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml
index f8e513756d32..ac8d24d3fec4 100644
--- a/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml
+++ b/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml
@@ -35,6 +35,7 @@ server_configs:
kv_cache_config:
dtype: 'fp8'
enable_block_reuse: false
+ use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.9
tokens_per_block: 64
cache_transceiver_config:
diff --git a/tests/scripts/perf-sanity/aggregated/dynamo_k25_thinking_fp4_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/dynamo_k25_thinking_fp4_blackwell.yaml
index 2233f5040482..50728425ebad 100644
--- a/tests/scripts/perf-sanity/aggregated/dynamo_k25_thinking_fp4_blackwell.yaml
+++ b/tests/scripts/perf-sanity/aggregated/dynamo_k25_thinking_fp4_blackwell.yaml
@@ -26,8 +26,8 @@ server_configs:
dtype: 'fp8'
free_gpu_memory_fraction: 0.75
cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
+ backend: NIXL
+ transceiver_runtime: PYTHON
max_tokens_in_buffer: 8448
client_configs:
- name: "con128_iter5_2k1k"
diff --git a/tests/scripts/perf-sanity/cache_transceiver_precheck/precheck_config.py b/tests/scripts/perf-sanity/cache_transceiver_precheck/precheck_config.py
index 497edae4377d..970a254abf9b 100644
--- a/tests/scripts/perf-sanity/cache_transceiver_precheck/precheck_config.py
+++ b/tests/scripts/perf-sanity/cache_transceiver_precheck/precheck_config.py
@@ -449,7 +449,7 @@ def resolve_plan(cfg, benchmark_mode="e2e"):
# True/False from the yaml wins; absent means "auto", which the
# driver resolves against the model class's manager preference at
# runtime, exactly like serving (_resolve_kv_cache_manager_v2_auto).
- plan[f"{role}_use_kv_cache_manager_v2"] = kv_cfg.get("use_kv_cache_manager_v2", "auto")
+ plan[f"{role}_use_kv_cache_manager_v2"] = kv_cfg.get("use_kv_cache_manager_v2", True)
plan["fingerprint"] = plan_fingerprint(plan)
return plan
diff --git a/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml
index bc277d40c127..a055fcfbb2d9 100644
--- a/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml
@@ -59,6 +59,7 @@ worker_config:
max_batch_size: 16
kv_cache_config:
enable_block_reuse: false
+ use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.85
moe_config:
backend: TRTLLM
@@ -82,6 +83,7 @@ worker_config:
max_batch_size: 16
kv_cache_config:
enable_block_reuse: false
+ use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.85
moe_config:
backend: TRTLLM
diff --git a/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_128k8k_con256_ctx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_128k8k_con256_ctx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml
index c3ac0ccb63a2..d109498ebdbe 100644
--- a/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_128k8k_con256_ctx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_128k8k_con256_ctx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml
@@ -67,6 +67,7 @@ worker_config:
max_batch_size: 16
kv_cache_config:
enable_block_reuse: false
+ use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.8
dtype: fp8
moe_config:
@@ -95,6 +96,7 @@ worker_config:
cuda_graph_config: null
kv_cache_config:
enable_block_reuse: false
+ use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.3
dtype: fp8
moe_config:
diff --git a/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml
index 02ac4618d977..e67c1a90ca82 100644
--- a/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml
@@ -66,6 +66,7 @@ worker_config:
max_batch_size: 256
kv_cache_config:
enable_block_reuse: false
+ use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.8
dtype: fp8
moe_config:
@@ -94,6 +95,7 @@ worker_config:
cuda_graph_config: null
kv_cache_config:
enable_block_reuse: false
+ use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.6
dtype: fp8
moe_config:
diff --git a/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py b/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py
index 03aeafb8b328..d1386439b730 100644
--- a/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py
+++ b/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py
@@ -488,7 +488,7 @@ def test_direct_cpp_wrapper_rejects_python_runtime_opt_in(monkeypatch):
def test_flag_unset_preserves_existing_backend_selection(monkeypatch):
- config = CacheTransceiverConfig(backend="UCX")
+ config = CacheTransceiverConfig(backend="UCX", transceiver_runtime="CPP")
expected = object()
constructor = Mock(return_value=expected)
monkeypatch.setattr(transceiver_module, "BindKvCacheTransceiver", constructor)
@@ -568,7 +568,7 @@ def test_cpp_runtime_keeps_cpp_mamba_manager(monkeypatch, runtime):
def test_flag_unset_preserves_libfabric_selection(monkeypatch):
monkeypatch.setenv(transceiver_module._NIXL_KVCACHE_BACKEND_ENV, "LIBFABRIC")
- config = CacheTransceiverConfig(backend="NIXL")
+ config = CacheTransceiverConfig(backend="NIXL", transceiver_runtime="CPP")
expected = object()
constructor = Mock(return_value=expected)
monkeypatch.setattr(transceiver_module, "BindKvCacheTransceiver", constructor)
@@ -590,7 +590,7 @@ def test_flag_unset_preserves_libfabric_selection(monkeypatch):
)
def test_flag_unset_preserves_legacy_backend_env(monkeypatch, selector, expected_backend):
monkeypatch.setenv(selector, "1")
- config = CacheTransceiverConfig(backend="DEFAULT")
+ config = CacheTransceiverConfig(backend="DEFAULT", transceiver_runtime="CPP")
constructor = Mock(return_value=object())
monkeypatch.setattr(transceiver_module, "BindKvCacheTransceiver", constructor)
@@ -608,7 +608,7 @@ def test_flag_unset_preserves_legacy_backend_env_precedence(monkeypatch):
"TRTLLM_USE_NIXL_KVCACHE",
):
monkeypatch.setenv(selector, "1")
- config = CacheTransceiverConfig(backend="DEFAULT")
+ config = CacheTransceiverConfig(backend="DEFAULT", transceiver_runtime="CPP")
monkeypatch.setattr(transceiver_module, "BindKvCacheTransceiver", Mock())
transceiver_module.create_kv_cache_transceiver(Mock(), Mock(), Mock(), Mock(), config)
diff --git a/tests/unittest/_torch/executor/kv_cache/test_kv_cache_estimation.py b/tests/unittest/_torch/executor/kv_cache/test_kv_cache_estimation.py
index 2d698874e9a8..452a5e27b6b1 100644
--- a/tests/unittest/_torch/executor/kv_cache/test_kv_cache_estimation.py
+++ b/tests/unittest/_torch/executor/kv_cache/test_kv_cache_estimation.py
@@ -1255,6 +1255,36 @@ def test_estimation_temporarily_uses_inferred_pool_sizing(
assert kv_cache_config.avg_seq_len == avg_seq_len
+@pytest.mark.parametrize(
+ ("is_v2", "cp_config", "expected_skip_est"),
+ [
+ (True, {}, True),
+ (False, {}, False),
+ (True, {"cp_type": "ring"}, False),
+ ],
+)
+def test_vanilla_attention_uses_capacity_fallback_for_v2(
+ is_v2: bool,
+ cp_config: dict,
+ expected_skip_est: bool,
+) -> None:
+ creator = object.__new__(KvCacheCreator)
+ creator._skip_est = False
+ creator._mapping = SimpleNamespace(cp_config=cp_config)
+ creator._model_engine = SimpleNamespace(
+ model=SimpleNamespace(
+ model_config=SimpleNamespace(
+ attn_backend="VANILLA",
+ is_encoder_decoder=False,
+ )
+ )
+ )
+ creator._is_kv_cache_manager_v2 = is_v2
+
+ assert creator.try_prepare_estimation() is False
+ assert creator._skip_est is expected_skip_est
+
+
@pytest.mark.parametrize(
("estimating_kv_cache", "expected_avg_seq_len"),
[(True, 2045), (False, 2055)],
diff --git a/tests/unittest/_torch/executor/kv_cache/test_mamba_cache_manager.py b/tests/unittest/_torch/executor/kv_cache/test_mamba_cache_manager.py
index 1f920c7b718c..a3c86a777a8c 100644
--- a/tests/unittest/_torch/executor/kv_cache/test_mamba_cache_manager.py
+++ b/tests/unittest/_torch/executor/kv_cache/test_mamba_cache_manager.py
@@ -843,6 +843,7 @@ def test_hybrid_cache_manager_factory_rejects_mixed_override_with_reuse(
KvCacheConfig(
enable_block_reuse=True,
mamba_state_config=MambaStateConfig(periodic_snapshot_interval=256),
+ use_kv_cache_manager_v2=False,
),
)
@@ -901,7 +902,10 @@ def test_hybrid_models_prefer_v2_and_python_transceiver(monkeypatch):
):
llm_args = TorchLlmArgs(
model="/tmp/dummy_model",
- cache_transceiver_config=CacheTransceiverConfig(backend="DEFAULT"),
+ kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"),
+ cache_transceiver_config=CacheTransceiverConfig(
+ backend="DEFAULT", transceiver_runtime="auto"
+ ),
)
_resolve_transceiver_runtime_auto(llm_args, model_cls)
_resolve_kv_cache_manager_v2_auto(llm_args, model_cls)
@@ -969,6 +973,7 @@ def test_kimi_without_v2_preference_uses_mixed_manager(
kv_cache_config=KvCacheConfig(
enable_block_reuse=False,
tokens_per_block=64,
+ use_kv_cache_manager_v2="auto",
),
)
resolved = _resolve_kv_cache_manager_v2_auto(llm_args, KimiLinearForCausalLM)
@@ -995,7 +1000,7 @@ def test_kimi_preferred_transceiver_runtime() -> None:
"cache_transceiver_config",
[
None,
- CacheTransceiverConfig(backend="NIXL"), # runtime left at 'auto'
+ CacheTransceiverConfig(backend="NIXL", transceiver_runtime="auto"),
CacheTransceiverConfig(backend="NIXL", transceiver_runtime="CPP"),
CacheTransceiverConfig(backend="UCX", transceiver_runtime="CPP"),
CacheTransceiverConfig(backend="UCX", transceiver_runtime="PYTHON"),
@@ -1031,7 +1036,10 @@ def test_kimi_disagg_python_nixl_routes_to_mixed_manager(
assert (
get_kv_cache_manager_cls(
_kimi_model_config(),
- KvCacheConfig(enable_block_reuse=False),
+ KvCacheConfig(
+ enable_block_reuse=False,
+ use_kv_cache_manager_v2=False,
+ ),
is_disagg=True,
cache_transceiver_config=CacheTransceiverConfig(
backend="NIXL", transceiver_runtime="PYTHON"
diff --git a/tests/unittest/_torch/modeling/test_modeling_gpt_oss.py b/tests/unittest/_torch/modeling/test_modeling_gpt_oss.py
index 197c9897242e..358d9ce0d066 100644
--- a/tests/unittest/_torch/modeling/test_modeling_gpt_oss.py
+++ b/tests/unittest/_torch/modeling/test_modeling_gpt_oss.py
@@ -63,9 +63,10 @@ def _resolve_gpt_oss_kv_cache_manager_v2(**llm_args_kwargs) -> bool:
return _resolve_kv_cache_manager_v2_auto(llm_args, GptOssForCausalLM)
-def test_gpt_oss_model_preference_selects_v2():
+def test_gpt_oss_auto_selects_model_preference():
"""GPT-OSS is VSWA, so "auto" resolves to KVCacheManagerV2."""
- assert _resolve_gpt_oss_kv_cache_manager_v2() is True
+ assert _resolve_gpt_oss_kv_cache_manager_v2(kv_cache_config=KvCacheConfig(
+ use_kv_cache_manager_v2="auto")) is True
@pytest.mark.parametrize("user_setting", [False, True])
diff --git a/tests/unittest/_torch/speculative/test_eagle3.py b/tests/unittest/_torch/speculative/test_eagle3.py
index 655571c9092e..1c2556646998 100644
--- a/tests/unittest/_torch/speculative/test_eagle3.py
+++ b/tests/unittest/_torch/speculative/test_eagle3.py
@@ -773,7 +773,8 @@ def test_eagle3_spec_decoding_stats(eagle3_one_model):
pytest.skip(f"Required models not found")
kv_cache_config = KvCacheConfig(enable_block_reuse=False,
- free_gpu_memory_fraction=0.6)
+ free_gpu_memory_fraction=0.6,
+ use_kv_cache_manager_v2=eagle3_one_model)
spec_config = Eagle3DecodingConfig(
max_draft_len=3,
speculative_model=eagle_model_dir,
@@ -863,8 +864,10 @@ def test_llama_eagle3_long_prompt(use_cuda_graph):
else:
cuda_graph_config = None
+ kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2=False)
llm_spec = LLM(model=target_model_dir,
speculative_config=spec_config,
+ kv_cache_config=kv_cache_config,
max_batch_size=1,
cuda_graph_config=cuda_graph_config,
disable_overlap_scheduler=True)
@@ -878,6 +881,7 @@ def test_llama_eagle3_long_prompt(use_cuda_graph):
llm_spec.shutdown()
llm_ref = LLM(model=target_model_dir,
+ kv_cache_config=kv_cache_config,
max_batch_size=1,
cuda_graph_config=None,
disable_overlap_scheduler=False)
@@ -1066,7 +1070,8 @@ def test_multi_eagle3(use_one_model: bool):
max_batch_size = 16
max_draft_len = 3
kv_cache_config = KvCacheConfig(enable_block_reuse=enable_block_reuse,
- free_gpu_memory_fraction=0.5)
+ free_gpu_memory_fraction=0.5,
+ use_kv_cache_manager_v2=use_one_model)
cuda_graph_config = CudaGraphConfig(
batch_sizes=[1]) if use_cuda_graph else None
diff --git a/tests/unittest/llmapi/test_async_llm.py b/tests/unittest/llmapi/test_async_llm.py
index 9468eedaa6d5..7c9f0b6f9826 100644
--- a/tests/unittest/llmapi/test_async_llm.py
+++ b/tests/unittest/llmapi/test_async_llm.py
@@ -142,7 +142,7 @@ async def test_async_llm_placement_api(setup_ray_cluster, monkeypatch):
@pytest.mark.asyncio
async def test_async_llm_reset_prefix_cache():
llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0")
- kv_cache_config = KvCacheConfig(enable_block_reuse=True)
+ kv_cache_config = KvCacheConfig(enable_block_reuse=True, use_kv_cache_manager_v2=False)
prompt = "The future of AI is " * 20
sampling_params = SamplingParams(temperature=0, max_tokens=5, return_perf_metrics=True)
diff --git a/tests/unittest/llmapi/test_llm_args.py b/tests/unittest/llmapi/test_llm_args.py
index dca3820ed3ee..b1e3a17854f8 100644
--- a/tests/unittest/llmapi/test_llm_args.py
+++ b/tests/unittest/llmapi/test_llm_args.py
@@ -853,10 +853,8 @@ def test_nvfp4_resolution_preserves_frozen_checkpoint_config(
with pytest.raises(AttributeError, match="instance is frozen"):
config.attn_backend = "TRTLLM"
- @pytest.mark.parametrize("explicit_auto", [False, True])
- def test_auto_uses_model_preference(self, explicit_auto):
- kv_cache_config = (KvCacheConfig(use_kv_cache_manager_v2="auto")
- if explicit_auto else KvCacheConfig())
+ def test_auto_uses_model_preference(self):
+ kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2="auto")
llm_args = TorchLlmArgs(model="/tmp/dummy_model",
kv_cache_config=kv_cache_config)
@@ -865,7 +863,10 @@ def test_auto_uses_model_preference(self, explicit_auto):
assert llm_args.kv_cache_config.use_kv_cache_manager_v2 is True
def test_auto_without_preference_falls_back_to_v1(self):
- llm_args = TorchLlmArgs(model="/tmp/dummy_model")
+ llm_args = TorchLlmArgs(
+ model="/tmp/dummy_model",
+ kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"),
+ )
_resolve_kv_cache_manager_v2_auto(llm_args)
@@ -882,6 +883,7 @@ def test_auto_without_preference_falls_back_to_v1(self):
def test_auto_v2_falls_back_for_incompatible_disagg(self, backend, runtime):
llm_args = TorchLlmArgs(
model="/tmp/dummy_model",
+ kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"),
cache_transceiver_config=CacheTransceiverConfig(
backend=backend, transceiver_runtime=runtime),
)
@@ -893,6 +895,7 @@ def test_auto_v2_falls_back_for_incompatible_disagg(self, backend, runtime):
def test_auto_v2_keeps_python_nixl_preference(self):
llm_args = TorchLlmArgs(
model="/tmp/dummy_model",
+ kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"),
cache_transceiver_config=CacheTransceiverConfig(
backend="NIXL", transceiver_runtime="PYTHON"),
)
@@ -1015,7 +1018,7 @@ def test_KvCacheConfig_declaration():
assert KvCacheConfig().kv_cache_event_hash_algo == "auto"
assert KvCacheConfig().block_reuse_config == BlockReuseConfig()
assert KvCacheConfig().enable_swa_scratch_reuse is False
- assert KvCacheConfig().use_kv_cache_manager_v2 == "auto"
+ assert KvCacheConfig().use_kv_cache_manager_v2 is True
assert KvCacheConfig(
use_kv_cache_manager_v2=True).use_kv_cache_manager_v2 is True
assert KvCacheConfig(
@@ -4375,15 +4378,17 @@ class TestTransceiverRuntimeAutoResolution:
"""Tests for the transceiver_runtime 'auto' selection mechanism."""
def _disagg_args(self, backend="NIXL", **cfg_kwargs):
+ cfg_kwargs.setdefault("transceiver_runtime", "auto")
return TorchLlmArgs(
model="/tmp/dummy_model",
+ kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"),
cache_transceiver_config=CacheTransceiverConfig(backend=backend,
**cfg_kwargs),
)
- def test_default_is_auto(self):
+ def test_default_is_python(self):
cfg = CacheTransceiverConfig(backend="NIXL")
- assert cfg.transceiver_runtime == "auto"
+ assert cfg.transceiver_runtime == "PYTHON"
def test_auto_no_model_preference_defaults_to_python(self) -> None:
"""'auto' with no model preference resolves to the Python transceiver."""
@@ -4391,11 +4396,9 @@ def test_auto_no_model_preference_defaults_to_python(self) -> None:
_resolve_transceiver_runtime_auto(args)
assert args.cache_transceiver_config.transceiver_runtime == "PYTHON"
- @pytest.mark.parametrize("explicit_auto", [False, True])
- def test_model_preference_adopted(self, explicit_auto):
- """Model preference applies whether 'auto' is implicit or explicit."""
- cfg_kwargs = {"transceiver_runtime": "auto"} if explicit_auto else {}
- args = self._disagg_args(**cfg_kwargs)
+ def test_explicit_auto_adopts_model_preference(self):
+ """An explicit 'auto' adopts the model preference."""
+ args = self._disagg_args()
_resolve_transceiver_runtime_auto(args, _PreferPythonTransceiverModel)
assert args.cache_transceiver_config.transceiver_runtime == "PYTHON"
@@ -4513,7 +4516,7 @@ def test_backend_none_is_noop(self):
)
_resolve_transceiver_runtime_auto(args, _PreferPythonTransceiverModel)
assert args.cache_transceiver_config.backend is None
- assert args.cache_transceiver_config.transceiver_runtime == "auto"
+ assert args.cache_transceiver_config.transceiver_runtime == "PYTHON"
def test_invalid_model_preference_raises(self):
@@ -4778,7 +4781,8 @@ def test_deepseek_resolves_auto_to_python_on_nixl(self) -> None:
DeepseekV3ForCausalLM
args = TorchLlmArgs(
model="/tmp/dummy_model",
- cache_transceiver_config=CacheTransceiverConfig(backend="NIXL"),
+ cache_transceiver_config=CacheTransceiverConfig(
+ backend="NIXL", transceiver_runtime="auto"),
)
cfg = self._pretrained_config(["DeepseekV3ForCausalLM"], "deepseek_v3")
_resolve_transceiver_runtime_auto(args, DeepseekV3ForCausalLM, cfg)
diff --git a/tests/unittest/llmapi/test_llm_pytorch.py b/tests/unittest/llmapi/test_llm_pytorch.py
index 5a5d80c99382..3c9f31a24549 100644
--- a/tests/unittest/llmapi/test_llm_pytorch.py
+++ b/tests/unittest/llmapi/test_llm_pytorch.py
@@ -65,6 +65,13 @@
# isort: on
+def _kv_cache_config_for_transceiver_runtime(
+ config: KvCacheConfig,
+ transceiver_runtime: Optional[str]) -> KvCacheConfig:
+ return config.model_copy(
+ update={"use_kv_cache_manager_v2": transceiver_runtime == "PYTHON"})
+
+
@force_ampere
@pytest.mark.parametrize("enable_chunked_prefill", [False, True])
@pytest.mark.part2
@@ -1124,7 +1131,8 @@ def test_llm_context_only_timed_out(transceiver_runtime):
# Python transceiver (V2) only supports NIXL/DEFAULT backends
backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX"
llm = LLM(model=llama_model_path,
- kv_cache_config=global_kvcache_config,
+ kv_cache_config=_kv_cache_config_for_transceiver_runtime(
+ global_kvcache_config, transceiver_runtime),
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
@@ -1219,9 +1227,10 @@ def test_llm_context_only_timed_out_kv_cache_exhausted(sender_future_timeout_ms,
enable_iter_req_stats=enable_iter_req_stats,
disable_overlap_scheduler=not use_overlap))
- kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.1,
- max_tokens=1000,
- enable_block_reuse=False)
+ kv_cache_config = _kv_cache_config_for_transceiver_runtime(
+ KvCacheConfig(free_gpu_memory_fraction=0.1,
+ max_tokens=1000,
+ enable_block_reuse=False), transceiver_runtime)
llm = LLM(model=llama_model_path,
kv_cache_config=kv_cache_config,
tensor_parallel_size=tp_size,
@@ -1308,7 +1317,8 @@ async def test_llm_disagg_gen_cancelled(transceiver_runtime):
# Python transceiver (V2) only supports NIXL/DEFAULT backends
backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX"
llm_ctx = LLM(model=llama_model_path,
- kv_cache_config=global_kvcache_config_no_reuse,
+ kv_cache_config=_kv_cache_config_for_transceiver_runtime(
+ global_kvcache_config_no_reuse, transceiver_runtime),
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
@@ -1317,7 +1327,8 @@ async def test_llm_disagg_gen_cancelled(transceiver_runtime):
**llm_args_extra)
llm_gen = LLM(model=llama_model_path,
- kv_cache_config=global_kvcache_config_no_reuse,
+ kv_cache_config=_kv_cache_config_for_transceiver_runtime(
+ global_kvcache_config_no_reuse, transceiver_runtime),
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
@@ -1488,7 +1499,8 @@ async def test_llm_disagg_streaming_gen_cancelled(transceiver_runtime):
# Python transceiver (V2) only supports NIXL/DEFAULT backends
backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX"
llm_ctx = LLM(model=llama_model_path,
- kv_cache_config=global_kvcache_config_no_reuse,
+ kv_cache_config=_kv_cache_config_for_transceiver_runtime(
+ global_kvcache_config_no_reuse, transceiver_runtime),
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
@@ -1497,7 +1509,8 @@ async def test_llm_disagg_streaming_gen_cancelled(transceiver_runtime):
**llm_args_extra)
llm_gen = LLM(model=llama_model_path,
- kv_cache_config=global_kvcache_config_no_reuse,
+ kv_cache_config=_kv_cache_config_for_transceiver_runtime(
+ global_kvcache_config_no_reuse, transceiver_runtime),
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
diff --git a/tests/unittest/llmapi/test_quickstart_advanced.py b/tests/unittest/llmapi/test_quickstart_advanced.py
index d9e9212721c2..3e33e93d0aec 100644
--- a/tests/unittest/llmapi/test_quickstart_advanced.py
+++ b/tests/unittest/llmapi/test_quickstart_advanced.py
@@ -44,9 +44,9 @@ def test_use_kv_cache_manager_v2_cli_values(cli_value: str, expected: str | bool
assert args.use_kv_cache_manager_v2 == expected
-def test_use_kv_cache_manager_v2_cli_default_is_auto() -> None:
+def test_use_kv_cache_manager_v2_cli_default_is_true() -> None:
parser = _MODULE.add_llm_args(argparse.ArgumentParser())
args = parser.parse_args(["--model_dir", "dummy-model"])
- assert args.use_kv_cache_manager_v2 == "auto"
+ assert args.use_kv_cache_manager_v2 is True
diff --git a/tests/unittest/others/test_cache_transceiver_precheck_config.py b/tests/unittest/others/test_cache_transceiver_precheck_config.py
index 561d3107a05a..e8c8f379ce08 100644
--- a/tests/unittest/others/test_cache_transceiver_precheck_config.py
+++ b/tests/unittest/others/test_cache_transceiver_precheck_config.py
@@ -289,9 +289,9 @@ def test_use_kv_cache_manager_v2_flags():
# Absent -> "auto" (the driver resolves it against the model's
# get_preferred_kv_cache_manager_version at runtime, like serving).
plan = pcfg.resolve_plan(_disagg_yaml())
- assert plan["ctx_use_kv_cache_manager_v2"] == "auto"
- assert plan["gen_use_kv_cache_manager_v2"] == "auto"
- assert pcfg.side_plan(plan, "ctx")["use_kv_cache_manager_v2"] == "auto"
+ assert plan["ctx_use_kv_cache_manager_v2"] is True
+ assert plan["gen_use_kv_cache_manager_v2"] is True
+ assert pcfg.side_plan(plan, "ctx")["use_kv_cache_manager_v2"] is True
# Explicit yaml values win, per side.
plan = pcfg.resolve_plan(
diff --git a/tests/unittest/others/test_kv_cache_transceiver.py b/tests/unittest/others/test_kv_cache_transceiver.py
index 1179365af005..7193a737fc0a 100644
--- a/tests/unittest/others/test_kv_cache_transceiver.py
+++ b/tests/unittest/others/test_kv_cache_transceiver.py
@@ -415,6 +415,7 @@ def test_cancel_request_in_transmission(attention_type):
kv_cache_manager_gen = create_kv_cache_manager(mapping, gen_kv_cache_dtype)
cache_transceiver_config = CacheTransceiverConfig(backend="DEFAULT",
+ transceiver_runtime="CPP",
max_tokens_in_buffer=512)
kv_cache_transceiver_ctx = create_kv_cache_transceiver(
@@ -486,6 +487,7 @@ def test_async_transfer_keeps_llm_request_alive():
kv_cache_manager_gen = create_kv_cache_manager(mapping, DataType.HALF)
cache_transceiver_config = CacheTransceiverConfig(backend="DEFAULT",
+ transceiver_runtime="CPP",
max_tokens_in_buffer=512)
transceiver_ctx = create_kv_cache_transceiver(mapping, dist,
kv_cache_manager_ctx,
@@ -612,7 +614,10 @@ def test_kv_transfer_timeout_warns_once_per_request(capfd):
kv_cache_manager_ctx = create_kv_cache_manager(mapping, DataType.HALF)
cache_transceiver_config = CacheTransceiverConfig(
- backend="DEFAULT", max_tokens_in_buffer=512, kv_transfer_timeout_ms=100)
+ backend="DEFAULT",
+ transceiver_runtime="CPP",
+ max_tokens_in_buffer=512,
+ kv_transfer_timeout_ms=100)
transceiver_ctx = create_kv_cache_transceiver(mapping, dist,
kv_cache_manager_ctx,
AttentionTypeCpp.DEFAULT,
@@ -661,6 +666,7 @@ def test_kv_transfer_timeout_silent_when_unset(capfd):
kv_cache_manager_ctx = create_kv_cache_manager(mapping, DataType.HALF)
cache_transceiver_config = CacheTransceiverConfig(backend="DEFAULT",
+ transceiver_runtime="CPP",
max_tokens_in_buffer=512)
transceiver_ctx = create_kv_cache_transceiver(mapping, dist,
kv_cache_manager_ctx,
@@ -698,6 +704,7 @@ def test_context_transfer_bounded_poll_keeps_request_in_progress(capfd):
cache_transceiver_config = CacheTransceiverConfig(
backend="DEFAULT",
+ transceiver_runtime="CPP",
max_tokens_in_buffer=512,
kv_transfer_timeout_ms=100,
kv_transfer_sender_future_timeout_ms=10)
From 2da184ad58b17496772eaa5f2787a391e3475ae9 Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Tue, 18 Aug 2026 06:42:32 -0700
Subject: [PATCH 02/16] [None][test] Skip known KVCM V2 failures
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/test_lists/waives.txt | 25 +++++++++++++++++++++++++
1 file changed, 25 insertions(+)
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index 49833d85fb38..de1f08bc17f9 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -26,6 +26,9 @@ accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-dp4-cutl
accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-ep4-cutlass-auto] SKIP (https://nvbugs/5596343)
accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-tp4-cutlass-auto] SKIP (https://nvbugs/5596343)
accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_chunked_prefill[cutlass-auto] SKIP (https://nvbugs/5596343)
+accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_bf16_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_fp8_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_nvfp4_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_fp8[latency] SKIP (https://nvbugs/6177390)
accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_fp8[throughput_latency] SKIP (https://nvbugs/6177390)
accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B::test_nvfp4[tep4_latency_moe_cutlass-torch_compile=True] SKIP (https://nvbugs/6561558)
@@ -225,6 +228,11 @@ full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::Tes
full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_eagle3_4gpus[v2_kv_cache-cutlass-one_model-overlap_scheduler] SKIP (https://nvbugs/6672360)
full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_eagle3_vswa_reuse_4gpus[v1_kv_cache-one_model] SKIP (https://nvbugs/6672360)
full:sm100/unittest/bindings SKIP (Disable for Blackwell)
+kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[guaranteed-chunked] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[max-util-chunked] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[no-overlap-chunked] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[offload-chunked] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[python-scheduler] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
llmapi/test_llm_examples.py::test_llmapi_speculative_decoding_eagle3 SKIP (https://nvbugs/6075431)
perf/test_perf_sanity.py::test_e2e[aggr_upload-gen_only_no_context-gb300_nemotron-ultra-v3-fp4_50k2k_con12_ctx1_dep4_gen6_tep4_eplb0_mtp6_ccb-NIXL] SKIP (https://nvbugs/6813584)
perf/test_perf_sanity.py::test_e2e[aggr_upload-gen_only_no_context-gb300_nemotron-ultra-v3-fp4_8k64k_con1_ctx1_dep4_gen1_tep4_eplb0_mtp5_ccb-NIXL] SKIP (https://nvbugs/6813584)
@@ -283,8 +291,20 @@ unittest/_torch/modeling/test_modeling_nemotron_h_multimodal.py::test_nemotron_n
unittest/_torch/modules/tests_lora_modules/test_nemotron_h_lora_sanity.py::TestNemotronHLoRA::test_lora_pp2_sanity SKIP (https://nvbugs/6428124)
unittest/_torch/moe/test_moe_backend.py::test_moe_backend[act=Relu2-e60_k4_h2048_i1408-seq=8-dtype=torch.bfloat16-backend=TRTLLM-quant=NVFP4-routing=Renormalize] SKIP (https://nvbugs/5989912)
unittest/_torch/multi_gpu/test_linear.py::test_row_linear_norm_fusion[2-hidden:16-seqlen:2] SKIP (https://nvbugs/6501404)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_epd_disagg_mm_hash_kv_cache_reuse[prompts0] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_epd_disagg_mm_hash_kv_cache_reuse[prompts0] SKIP (https://nvbugs/6812085)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_epd_disagg_mm_hash_kv_cache_reuse[prompts1] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_partial_uuids[uuids0-expected_patterns0] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_partial_uuids[uuids1-expected_patterns1] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_partial_uuids[uuids2-expected_patterns2] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_reuse[prompts1-2] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_uuid[False-hex] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_uuid[True-uuid] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_uuid_multiple_prompts SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_very_long_uuid SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (https://nvbugs/6813129)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_with_block_reuse SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_single_request_chat_multiple_images[pd_disagg-qwen3_2b] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/ray_orchestrator/multi_gpu/test_llm_update_weights_multi_gpu.py::test_llm_update_weights_nemotron_h SKIP (https://nvbugs/6729495)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[False] SKIP (https://nvbugs/6535767)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[True] SKIP (https://nvbugs/6535767)
@@ -302,11 +322,16 @@ unittest/bindings/test_transfer_agent_bindings.py::TestMooncakeFunctionalTransfe
unittest/executor/test_rpc.py::TestRpcCorrectness::test_incremental_task_async SKIP (https://nvbugs/5741476)
unittest/executor/test_rpc_proxy.py SKIP (https://nvbugs/5605741)
unittest/executor/test_rpc_worker.py SKIP (https://nvbugs/5605741)
+unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_image SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_image_streaming SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_rgba_image SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/llmapi/test_llm_multi_gpu_pytorch.py -m "gpu2" SKIP (https://nvbugs/6428092)
unittest/llmapi/test_llm_multi_gpu_pytorch.py::test_llm_get_stats_pp2[False-False-True] SKIP (https://nvbugs/6432826)
unittest/llmapi/test_llm_pytorch.py::test_gqa_nemo_lora[None] SKIP (https://nvbugs/6162504)
unittest/llmapi/test_llm_pytorch.py::test_gqa_nemo_lora[cuda_graph_config0] SKIP (https://nvbugs/6162504)
unittest/llmapi/test_memory_profiling.py::test_profile_kvcache SKIP (https://nvbugs/5580781)
+unittest/tools/test_layer_wise_benchmarks.py::test_nemotron_gen_dep[1] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/tools/test_layer_wise_benchmarks.py::test_qwen3_next_gen_tep[1] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/usage/test_llmapi_config_telemetry_docs.py::test_build_capture_manifest_matches_committed_golden SKIP (https://nvbugs/6811958)
verl/test_verl_cases.py::test_async_generate SKIP (https://nvbugs/6683838)
verl/test_verl_cases.py::test_async_memory_management SKIP (https://nvbugs/6683838)
From 64bb58ca99045e6312ea6b268cf55ad890b20c92 Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Tue, 18 Aug 2026 19:50:33 -0700
Subject: [PATCH 03/16] [None][test] Skip latest KVCM V2 failures
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/test_lists/waives.txt | 3 +++
1 file changed, 3 insertions(+)
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index de1f08bc17f9..1a5980ccaac6 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -29,6 +29,8 @@ accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_chunked_prefill[cutlass-au
accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_bf16_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_fp8_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_nvfp4_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard[overlap_scheduler=False] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard[overlap_scheduler=True] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_fp8[latency] SKIP (https://nvbugs/6177390)
accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_fp8[throughput_latency] SKIP (https://nvbugs/6177390)
accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B::test_nvfp4[tep4_latency_moe_cutlass-torch_compile=True] SKIP (https://nvbugs/6561558)
@@ -304,6 +306,7 @@ unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_very_long_uuid SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (https://nvbugs/6813129)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_with_block_reuse SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_with_image_input[True-model_dir0] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_single_request_chat_multiple_images[pd_disagg-qwen3_2b] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/ray_orchestrator/multi_gpu/test_llm_update_weights_multi_gpu.py::test_llm_update_weights_nemotron_h SKIP (https://nvbugs/6729495)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[False] SKIP (https://nvbugs/6535767)
From a5fb9b2bb08aa23312199b878d0e61278cedceda Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Wed, 19 Aug 2026 00:23:58 -0700
Subject: [PATCH 04/16] [None][test] Skip concurrent KVCM V2 OOM
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/test_lists/waives.txt | 1 +
1 file changed, 1 insertion(+)
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index 1a5980ccaac6..8ac343ae87c3 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -304,6 +304,7 @@ unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_uuid[True-uuid] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_uuid_multiple_prompts SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_very_long_uuid SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (https://nvbugs/6813129)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_with_block_reuse SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_with_image_input[True-model_dir0] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
From 3db41cbde18a3422235a54adddcf3685d2cbd78e Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Mon, 14 Sep 2026 19:54:14 -0700
Subject: [PATCH 05/16] [None][test] Re-enable fixed and rerunnable KVCM V2
tests
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/test_lists/waives.txt | 18 ------------------
1 file changed, 18 deletions(-)
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index 8ac343ae87c3..3c4a86dd6600 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -26,11 +26,6 @@ accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-dp4-cutl
accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-ep4-cutlass-auto] SKIP (https://nvbugs/5596343)
accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_4gpus[v2_kv_cache-tp4-cutlass-auto] SKIP (https://nvbugs/5596343)
accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_w4_chunked_prefill[cutlass-auto] SKIP (https://nvbugs/5596343)
-accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_bf16_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_fp8_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-accuracy/test_llm_api_pytorch.py::TestLagunaXS_2_1::test_nvfp4_dflash SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard[overlap_scheduler=False] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-accuracy/test_llm_api_pytorch.py::TestLlama3_1_8BInstruct::test_pard[overlap_scheduler=True] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_fp8[latency] SKIP (https://nvbugs/6177390)
accuracy/test_llm_api_pytorch.py::TestQwen3_235B_A22B::test_fp8[throughput_latency] SKIP (https://nvbugs/6177390)
accuracy/test_llm_api_pytorch.py::TestQwen3_30B_A3B::test_nvfp4[tep4_latency_moe_cutlass-torch_compile=True] SKIP (https://nvbugs/6561558)
@@ -230,11 +225,6 @@ full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::Tes
full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_eagle3_4gpus[v2_kv_cache-cutlass-one_model-overlap_scheduler] SKIP (https://nvbugs/6672360)
full:RTX_PRO_6000_Blackwell_Server_Edition/accuracy/test_llm_api_pytorch.py::TestGPTOSS::test_eagle3_vswa_reuse_4gpus[v1_kv_cache-one_model] SKIP (https://nvbugs/6672360)
full:sm100/unittest/bindings SKIP (Disable for Blackwell)
-kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[guaranteed-chunked] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[max-util-chunked] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[no-overlap-chunked] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[offload-chunked] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[python-scheduler] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
llmapi/test_llm_examples.py::test_llmapi_speculative_decoding_eagle3 SKIP (https://nvbugs/6075431)
perf/test_perf_sanity.py::test_e2e[aggr_upload-gen_only_no_context-gb300_nemotron-ultra-v3-fp4_50k2k_con12_ctx1_dep4_gen6_tep4_eplb0_mtp6_ccb-NIXL] SKIP (https://nvbugs/6813584)
perf/test_perf_sanity.py::test_e2e[aggr_upload-gen_only_no_context-gb300_nemotron-ultra-v3-fp4_8k64k_con1_ctx1_dep4_gen1_tep4_eplb0_mtp5_ccb-NIXL] SKIP (https://nvbugs/6813584)
@@ -306,9 +296,6 @@ unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_very_long_uuid SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (https://nvbugs/6813129)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_with_block_reuse SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_with_image_input[True-model_dir0] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_single_request_chat_multiple_images[pd_disagg-qwen3_2b] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/ray_orchestrator/multi_gpu/test_llm_update_weights_multi_gpu.py::test_llm_update_weights_nemotron_h SKIP (https://nvbugs/6729495)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[False] SKIP (https://nvbugs/6535767)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[True] SKIP (https://nvbugs/6535767)
@@ -326,16 +313,11 @@ unittest/bindings/test_transfer_agent_bindings.py::TestMooncakeFunctionalTransfe
unittest/executor/test_rpc.py::TestRpcCorrectness::test_incremental_task_async SKIP (https://nvbugs/5741476)
unittest/executor/test_rpc_proxy.py SKIP (https://nvbugs/5605741)
unittest/executor/test_rpc_worker.py SKIP (https://nvbugs/5605741)
-unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_image SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_image_streaming SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_rgba_image SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/llmapi/test_llm_multi_gpu_pytorch.py -m "gpu2" SKIP (https://nvbugs/6428092)
unittest/llmapi/test_llm_multi_gpu_pytorch.py::test_llm_get_stats_pp2[False-False-True] SKIP (https://nvbugs/6432826)
unittest/llmapi/test_llm_pytorch.py::test_gqa_nemo_lora[None] SKIP (https://nvbugs/6162504)
unittest/llmapi/test_llm_pytorch.py::test_gqa_nemo_lora[cuda_graph_config0] SKIP (https://nvbugs/6162504)
unittest/llmapi/test_memory_profiling.py::test_profile_kvcache SKIP (https://nvbugs/5580781)
-unittest/tools/test_layer_wise_benchmarks.py::test_nemotron_gen_dep[1] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/tools/test_layer_wise_benchmarks.py::test_qwen3_next_gen_tep[1] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/usage/test_llmapi_config_telemetry_docs.py::test_build_capture_manifest_matches_committed_golden SKIP (https://nvbugs/6811958)
verl/test_verl_cases.py::test_async_generate SKIP (https://nvbugs/6683838)
verl/test_verl_cases.py::test_async_memory_management SKIP (https://nvbugs/6683838)
From ca1d4fc93becd5c5d40df4b4430afd3ec5813188 Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Mon, 14 Sep 2026 21:16:52 -0700
Subject: [PATCH 06/16] test: remove redundant V1 disaggregation coverage
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
.../accuracy/test_disaggregated_serving.py | 2 -
...ig_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml | 39 --------------
...ig_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml | 23 --------
.../defs/disaggregated/test_disaggregated.py | 24 ---------
.../test_lists/qa/llm_function_core.txt | 1 -
.../test_lists/test-db/l0_dgx_h100.yml | 1 -
tests/unittest/llmapi/test_llm_pytorch.py | 52 ++++++-------------
7 files changed, 16 insertions(+), 126 deletions(-)
delete mode 100644 tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml
delete mode 100644 tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml
diff --git a/tests/integration/defs/accuracy/test_disaggregated_serving.py b/tests/integration/defs/accuracy/test_disaggregated_serving.py
index e82d48d1fafa..7e3512721ed3 100644
--- a/tests/integration/defs/accuracy/test_disaggregated_serving.py
+++ b/tests/integration/defs/accuracy/test_disaggregated_serving.py
@@ -914,7 +914,6 @@ def test_auto_dtype_with_helix(self, comms_medium, cuda_graph_config,
# adopted verbatim and fail at creation on a non-NIXL backend.
"cache_transceiver_config": {
"backend": "DEFAULT",
- "transceiver_runtime": "CPP",
"max_tokens_in_buffer": 8192,
"transceiver_runtime": "CPP",
},
@@ -936,7 +935,6 @@ def test_auto_dtype_with_helix(self, comms_medium, cuda_graph_config,
"cuda_graph_config": cuda_graph_config,
"cache_transceiver_config": {
"backend": "DEFAULT",
- "transceiver_runtime": "CPP",
"max_tokens_in_buffer": 8192,
"transceiver_runtime": "CPP",
},
diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml
deleted file mode 100644
index 18897a334314..000000000000
--- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml
+++ /dev/null
@@ -1,39 +0,0 @@
-# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
-# SPDX-License-Identifier: Apache-2.0
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-# http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-
-hostname: localhost
-model: DeepSeek-V3-Lite/fp8
-free_gpu_memory_fraction: 0.1
-backend: pytorch
-cuda_graph_config: null
-disable_overlap_scheduler: true
-context_servers:
- num_instances: 1
- tensor_parallel_size: 1
- pipeline_parallel_size: 1
- kv_cache_config:
- use_kv_cache_manager_v2: false
- cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
-generation_servers:
- num_instances: 1
- tensor_parallel_size: 1
- pipeline_parallel_size: 1
- kv_cache_config:
- use_kv_cache_manager_v2: false
- cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml
deleted file mode 100644
index cff98e854555..000000000000
--- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml
+++ /dev/null
@@ -1,23 +0,0 @@
-hostname: localhost
-model: DeepSeek-V3-Lite/fp8
-free_gpu_memory_fraction: 0.25
-backend: pytorch
-disable_overlap_scheduler: true
-context_servers:
- num_instances: 1
- tensor_parallel_size: 2
- pipeline_parallel_size: 1
- kv_cache_config:
- use_kv_cache_manager_v2: false
- cache_transceiver_config:
- backend: MPI
- transceiver_runtime: CPP
-generation_servers:
- num_instances: 1
- tensor_parallel_size: 2
- pipeline_parallel_size: 1
- kv_cache_config:
- use_kv_cache_manager_v2: false
- cache_transceiver_config:
- backend: MPI
- transceiver_runtime: CPP
diff --git a/tests/integration/defs/disaggregated/test_disaggregated.py b/tests/integration/defs/disaggregated/test_disaggregated.py
index 163f05c7358b..24e3f4d4cf45 100644
--- a/tests/integration/defs/disaggregated/test_disaggregated.py
+++ b/tests/integration/defs/disaggregated/test_disaggregated.py
@@ -346,8 +346,6 @@ def get_test_config(test_desc, example_dir, test_root):
f"{test_configs_root}/disagg_config_ctxpp4_genpp4.yaml",
"ctxpp4_gentp4":
f"{test_configs_root}/disagg_config_ctxpp4_gentp4.yaml",
- "deepseek_v3_lite_fp8_mpi":
- f"{test_configs_root}/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml",
"deepseek_v3_lite_fp8_nixl":
f"{test_configs_root}/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_nixl.yaml",
"deepseek_v3_lite_fp8_tp1":
@@ -1800,28 +1798,6 @@ def test_disaggregated_ctxpp4_gentp4(disaggregated_test_root, llm_venv,
cwd=llm_venv.get_working_directory())
-@skip_no_hopper
-@pytest.mark.skip_less_device(4)
-@pytest.mark.skip(
- reason="MPI cache transceiver requires shared MPI process group, "
- "incompatible with service discovery which launches separate subprocesses")
-@pytest.mark.parametrize("deepseek_v3_model_root", ['DeepSeek-V3-Lite-fp8'],
- indirect=True)
-def test_disaggregated_deepseek_v3_lite_fp8_mpi(disaggregated_test_root,
- disaggregated_example_root,
- llm_venv,
- deepseek_v3_model_root):
- setup_model_symlink(llm_venv, deepseek_v3_model_root,
- "DeepSeek-V3-Lite/fp8")
- env = llm_venv._new_env.copy()
- env["TRTLLM_USE_MPI_KVCACHE"] = "1"
- run_disaggregated_test(disaggregated_example_root,
- "deepseek_v3_lite_fp8_mpi",
- env=env,
- model_path=deepseek_v3_model_root,
- cwd=llm_venv.get_working_directory())
-
-
@skip_no_hopper
@pytest.mark.parametrize("deepseek_v3_model_root", ['DeepSeek-V3-Lite-fp8'],
indirect=True)
diff --git a/tests/integration/test_lists/qa/llm_function_core.txt b/tests/integration/test_lists/qa/llm_function_core.txt
index 798a9e4da53c..e014ba906f89 100644
--- a/tests/integration/test_lists/qa/llm_function_core.txt
+++ b/tests/integration/test_lists/qa/llm_function_core.txt
@@ -695,7 +695,6 @@ disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_att
disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxpp2_gentp2_one_mtp[DeepSeek-V3-Lite-fp8]
disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxtp2ep2pp2_gentp4_one_mtp_block_reuse[DeepSeek-V3-Lite-fp8]
disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxtp2ep2pp2_gentp4_one_mtp_block_reuse_long_prompt[DeepSeek-V3-Lite-fp8]
-disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_mpi[DeepSeek-V3-Lite-fp8]
disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_nixl[DeepSeek-V3-Lite-fp8]
disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_overlap_cuda_graph[DeepSeek-V3-Lite-fp8]
disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1_attention_dp_overlap_one_mtp[DeepSeek-V3-Lite-fp8]
diff --git a/tests/integration/test_lists/test-db/l0_dgx_h100.yml b/tests/integration/test_lists/test-db/l0_dgx_h100.yml
index 7279dc0b1a7d..edb497c561e2 100644
--- a/tests/integration/test_lists/test-db/l0_dgx_h100.yml
+++ b/tests/integration/test_lists/test-db/l0_dgx_h100.yml
@@ -198,7 +198,6 @@ l0_dgx_h100:
- accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales_cuda_graph_padding_4gpus[attention_dp=True-mtp_nextn=0]
- accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales_cuda_graph_padding_4gpus[attention_dp=True-mtp_nextn=2]
- accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_guided_decoding_4gpus[xgrammar-mtp_nextn=2]
- - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_mpi[DeepSeek-V3-Lite-fp8]
- disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_nixl[DeepSeek-V3-Lite-fp8]
- disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_attention_dp[DeepSeek-V3-Lite-fp8]
- disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_attention_dp_overlap[DeepSeek-V3-Lite-fp8]
diff --git a/tests/unittest/llmapi/test_llm_pytorch.py b/tests/unittest/llmapi/test_llm_pytorch.py
index 3c9f31a24549..e2ed6c006473 100644
--- a/tests/unittest/llmapi/test_llm_pytorch.py
+++ b/tests/unittest/llmapi/test_llm_pytorch.py
@@ -65,13 +65,6 @@
# isort: on
-def _kv_cache_config_for_transceiver_runtime(
- config: KvCacheConfig,
- transceiver_runtime: Optional[str]) -> KvCacheConfig:
- return config.model_copy(
- update={"use_kv_cache_manager_v2": transceiver_runtime == "PYTHON"})
-
-
@force_ampere
@pytest.mark.parametrize("enable_chunked_prefill", [False, True])
@pytest.mark.part2
@@ -1115,7 +1108,7 @@ async def test_llm_rpc_get_stats_async():
@pytest.mark.threadleak(enabled=False)
@pytest.mark.part0
@skip_ray
-@pytest.mark.parametrize("transceiver_runtime", [None, "PYTHON"])
+@pytest.mark.parametrize("transceiver_runtime", ["PYTHON"])
def test_llm_context_only_timed_out(transceiver_runtime):
tp_size = 1
use_overlap = False
@@ -1128,11 +1121,9 @@ def test_llm_context_only_timed_out(transceiver_runtime):
enable_iter_req_stats=enable_iter_req_stats,
disable_overlap_scheduler=not use_overlap))
- # Python transceiver (V2) only supports NIXL/DEFAULT backends
- backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX"
+ backend = "NIXL"
llm = LLM(model=llama_model_path,
- kv_cache_config=_kv_cache_config_for_transceiver_runtime(
- global_kvcache_config, transceiver_runtime),
+ kv_cache_config=global_kvcache_config,
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
@@ -1207,15 +1198,11 @@ def test_llm_context_only_timed_out(transceiver_runtime):
@pytest.mark.part0
@skip_ray
@pytest.mark.parametrize("sender_future_timeout_ms", [100, 1000])
-@pytest.mark.parametrize("backend", ["NIXL", "UCX"])
-@pytest.mark.parametrize("transceiver_runtime", [None, "PYTHON"])
+@pytest.mark.parametrize("backend", ["NIXL"])
+@pytest.mark.parametrize("transceiver_runtime", ["PYTHON"])
def test_llm_context_only_timed_out_kv_cache_exhausted(sender_future_timeout_ms,
backend,
transceiver_runtime):
- # Python transceiver (V2) only supports NIXL/DEFAULT backends
- if transceiver_runtime == "PYTHON" and backend == "UCX":
- pytest.skip("Python transceiver (V2) does not support UCX backend")
-
tp_size = 1
use_overlap = False
enable_iter_req_stats = False
@@ -1227,10 +1214,9 @@ def test_llm_context_only_timed_out_kv_cache_exhausted(sender_future_timeout_ms,
enable_iter_req_stats=enable_iter_req_stats,
disable_overlap_scheduler=not use_overlap))
- kv_cache_config = _kv_cache_config_for_transceiver_runtime(
- KvCacheConfig(free_gpu_memory_fraction=0.1,
- max_tokens=1000,
- enable_block_reuse=False), transceiver_runtime)
+ kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.1,
+ max_tokens=1000,
+ enable_block_reuse=False)
llm = LLM(model=llama_model_path,
kv_cache_config=kv_cache_config,
tensor_parallel_size=tp_size,
@@ -1301,7 +1287,7 @@ def test_llm_context_only_timed_out_kv_cache_exhausted(sender_future_timeout_ms,
@pytest.mark.private_mpi_session
@pytest.mark.timeout(600)
@pytest.mark.asyncio
-@pytest.mark.parametrize("transceiver_runtime", [None, "PYTHON"])
+@pytest.mark.parametrize("transceiver_runtime", ["PYTHON"])
async def test_llm_disagg_gen_cancelled(transceiver_runtime):
tp_size = 1
use_overlap = False
@@ -1314,11 +1300,9 @@ async def test_llm_disagg_gen_cancelled(transceiver_runtime):
enable_iter_req_stats=enable_iter_req_stats,
disable_overlap_scheduler=not use_overlap))
- # Python transceiver (V2) only supports NIXL/DEFAULT backends
- backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX"
+ backend = "NIXL"
llm_ctx = LLM(model=llama_model_path,
- kv_cache_config=_kv_cache_config_for_transceiver_runtime(
- global_kvcache_config_no_reuse, transceiver_runtime),
+ kv_cache_config=global_kvcache_config_no_reuse,
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
@@ -1327,8 +1311,7 @@ async def test_llm_disagg_gen_cancelled(transceiver_runtime):
**llm_args_extra)
llm_gen = LLM(model=llama_model_path,
- kv_cache_config=_kv_cache_config_for_transceiver_runtime(
- global_kvcache_config_no_reuse, transceiver_runtime),
+ kv_cache_config=global_kvcache_config_no_reuse,
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
@@ -1483,7 +1466,7 @@ async def await_and_record(output, req_id: int):
@pytest.mark.part0
@skip_ray
@pytest.mark.asyncio
-@pytest.mark.parametrize("transceiver_runtime", [None, "PYTHON"])
+@pytest.mark.parametrize("transceiver_runtime", ["PYTHON"])
async def test_llm_disagg_streaming_gen_cancelled(transceiver_runtime):
tp_size = 1
use_overlap = False
@@ -1496,11 +1479,9 @@ async def test_llm_disagg_streaming_gen_cancelled(transceiver_runtime):
enable_iter_req_stats=enable_iter_req_stats,
disable_overlap_scheduler=not use_overlap))
- # Python transceiver (V2) only supports NIXL/DEFAULT backends
- backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX"
+ backend = "NIXL"
llm_ctx = LLM(model=llama_model_path,
- kv_cache_config=_kv_cache_config_for_transceiver_runtime(
- global_kvcache_config_no_reuse, transceiver_runtime),
+ kv_cache_config=global_kvcache_config_no_reuse,
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
@@ -1509,8 +1490,7 @@ async def test_llm_disagg_streaming_gen_cancelled(transceiver_runtime):
**llm_args_extra)
llm_gen = LLM(model=llama_model_path,
- kv_cache_config=_kv_cache_config_for_transceiver_runtime(
- global_kvcache_config_no_reuse, transceiver_runtime),
+ kv_cache_config=global_kvcache_config_no_reuse,
tensor_parallel_size=tp_size,
cache_transceiver_config=CacheTransceiverConfig(
backend=backend,
From d3929e426b76401512db2cfd4faa53a95a70b00d Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Tue, 15 Sep 2026 01:11:03 -0700
Subject: [PATCH 07/16] [None][fix] Use default KV cache manager for aggregated
DeepSeek V3.2
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
.../dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml | 5 -----
1 file changed, 5 deletions(-)
diff --git a/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml
index ac8d24d3fec4..6a0a0c9c2bc1 100644
--- a/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml
+++ b/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml
@@ -35,13 +35,8 @@ server_configs:
kv_cache_config:
dtype: 'fp8'
enable_block_reuse: false
- use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.9
tokens_per_block: 64
- cache_transceiver_config:
- backend: UCX
- transceiver_runtime: CPP
- max_tokens_in_buffer: 120000
client_configs:
- name: "con4_iter10_8k1k"
concurrency: 4
From 7caf188784973706c2ceda5194758a16cfa678f8 Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Tue, 15 Sep 2026 04:17:40 -0700
Subject: [PATCH 08/16] [None][test] Pin v1 disagg metrics and update disagg
coverage notes
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
pr-17495-disagg-cpp-by-file.md | 93 +++++++++
pr-17495-transceiver.html | 184 ++++++++++++++++++
.../test_configs/disagg_config_metrics.yaml | 12 +-
.../test_disaggregated_single_gpu.py | 4 +-
4 files changed, 286 insertions(+), 7 deletions(-)
create mode 100644 pr-17495-disagg-cpp-by-file.md
create mode 100644 pr-17495-transceiver.html
diff --git a/pr-17495-disagg-cpp-by-file.md b/pr-17495-disagg-cpp-by-file.md
new file mode 100644
index 000000000000..aaaf653b9d3d
--- /dev/null
+++ b/pr-17495-disagg-cpp-by-file.md
@@ -0,0 +1,93 @@
+
+
+# PR #17495 — Disagg 兼容设置与后续讨论
+
+[PR #17495](https://github.com/NVIDIA/TensorRT-LLM/pull/17495) 将 KVCM V2 设为默认。为保留现有 disagg 功能和回归覆盖,PR 对部分文件显式指定了 CPP runtime 或 KVCM V1。
+
+本 PR 以 CI 通过为合入前提,保留当前必要的 WAR(兼容设置),包括 AutoDeploy 的 CPP/V1 指定。AutoDeploy 已确认继续使用 V1,无需额外 fix。
+
+本文按文件列出当前处理和待讨论项,供 transceiver team review。第 2–3 节涉及的迁移和功能扩展取决于实际需求与已有覆盖,放在后续 PR 处理。已有等价 V2 coverage 的重复 V1 测试已在本 PR 删除,不列入后续跟进项。
+
+配置与支持范围:
+
+- **Transport:**团队支持 NIXL-based 方案,NIXL+CPP 在这一范围内。直接 UCX/MPI backend、NIXL 使用的 UCX plugin 和 MPI 启动设施属于不同层面的配置。
+- **Manager 与 runtime:**下文 V1/V2 指 KVCM 版本,CPP/Python 指 transceiver runtime。
+- **Helix:**目前按具体需求提供支持,已有实现对应 Kimi K3。是否扩展到 DeepSeek/Qwen 等模型取决于后续需求。
+
+## 1. AutoDeploy:保留 V1 及本 PR 的 CI WAR
+
+AutoDeploy 继续使用 V1 manager 和 CPP runtime。本 PR 保留显式 CPP/V1 WAR,使默认切到 V2 后仍走现有执行路径,并由本 PR 的 CI 验证。这部分无需修复或迁移。
+
+| 文件 | 本 PR 保留的 CI WAR |
+|---|---|
+| [examples/auto_deploy/model_registry/configs/disagg_ctx.yaml][ad_ctx] | `backend: DEFAULT` + `transceiver_runtime: CPP`;manager 由 AutoDeploy factory 创建为 V1。 |
+| [examples/auto_deploy/model_registry/configs/disagg_gen.yaml][ad_gen] | `backend: DEFAULT` + `transceiver_runtime: CPP`;manager 由 AutoDeploy factory 创建为 V1。 |
+| [tests/integration/defs/disaggregated/test_ad_disagg.py][ad_test] | 显式 V1+CPP。 |
+| [tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py][ad_serve] | 显式 V1+CPP。 |
+| [tests/unittest/auto_deploy/singlegpu/smoke/test_disagg.py][ad_smoke] | 显式 V1+CPP。 |
+
+## 2. Helix 与 Nemotron:支持范围及测试取舍
+
+Helix 的实现针对 Kimi K3 的具体需求。下列 DeepSeek/Qwen case 在本 PR 保留 CPP/V1 WAR;其他模型的支持扩展由后续需求决定。
+
+| 文件 | 本 PR 保留的设置与覆盖 | 后续讨论点 |
+|---|---|---|
+| [tests/integration/defs/accuracy/test_disaggregated_serving.py][accuracy] | DeepSeek/Qwen Helix 的 V1+CPP;launcher 通过环境变量选择 UCX | 若出现对应模型的支持需求,以同拓扑的精度及 overlap 测试评估 NIXL/V2 方案。 |
+| [tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml][helix_yaml] | 为既有 UCX+CPP 补显式 V1;保留 CTX TP2 → GEN TP1/CP2 | NIXL/V2 支持按后续需求评估,当前保留原有拓扑覆盖。 |
+| [tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml][nemotron_ucx] | 为既有 UCX+CPP 补显式 V1;保留 hybrid 模型性能场景 | 删除或迁移到 NIXL,取决于这条 Hopper workload 的维护需求与覆盖价值。若保留,再确定 NIXL 下的 manager/runtime 组合。 |
+
+Nemotron 这份配置仍由 [perf 测试自动收集][perf_collector]。目前未找到同 H200、TP2、8k/1k、concurrency 64 的等价 NIXL 配置,删除或迁移尚未确定。本 PR 保留现有 WAR。
+
+迁移方案的覆盖比较以原模型、拓扑、workload 和断言为基准。`fifo_v2` 表示通信实现;KVCM V2 的覆盖取决于 manager 配置和实际运行路径。
+
+## 3. C++ 专属接口与指标:NIXL 下的可选方案
+
+| 文件 | 当前依赖 / 限制 | 后续讨论点 |
+|---|---|---|
+| [tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py][single_gpu] | arbitrary-transfer 测试使用 C++ serialized `DataTransceiverState`;Python `get_context_state()` 尚未实现。本 PR 为既有 CPP case 补 V1。 | NIXL+CPP 是候选保留方案。若已有 Python 等价接口计划,对应覆盖包括成功传输和缺块错误。 |
+| [tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml][llama4] | 本 PR 显式指定 CPP/V1,保留 128k input 与 2048-token C++ transfer buffer 的溢出回归。 | NIXL+CPP 是候选迁移方案,覆盖比较关注原 buffer 路径及溢出触发条件。 |
+| [tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml][metrics] | 本 PR 使用 runtime/manager `auto`;调用方 UCX 环境变量使其选择 CPP/V1。现有 timing metrics 与 send/recv CSV 缺少等价 Python 验证。 | 待确定保留现有指标所需的 NIXL runtime。迁移涉及调用方和配置;Python 方案的覆盖以现有指标及 CSV 断言为基准。 |
+
+上述候选方案尚未验证。覆盖是否等价取决于原断言是否通过,以及接口、buffer 和指标路径是否仍实际执行。
+
+## 4. 保留的 C++ 专项回归
+
+以下测试验证 C++ 本身的行为,本 PR 通过显式 CPP 指定固定测试对象。这部分保留为专项回归,无 Python 功能补齐事项。
+
+| 文件 | 保留原因 |
+|---|---|
+| [tests/unittest/others/test_kv_cache_transceiver.py][binding_tests] | 五处 CPP 指定分别覆盖 C++ cancel 状态、shared_ptr/GC 生命周期、timeout warning 去重、warning 开关和 bounded polling;无 legacy selector 时使用 NIXL+CPP。 |
+| [tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py][cancel_gate] | 四处 CPP 指定固定 backend/配置选择行为,包含 NIXL+LIBFABRIC 和 legacy selector 优先级。legacy selector 测试随对应配置接口退役再清理。 |
+
+当前待讨论项集中在 Nemotron 的覆盖取舍,以及第 3 节的 NIXL 接口与指标方案。后续若确认等价覆盖,可据此评估撤回对应 WAR,并用 CI 验证。本 PR 保留当前必要的 WAR,以 CI 通过后合入为目标。
+
+代码核查版本:`5eed11ca57`,下方文件链接固定到该提交。AutoDeploy 和 Helix 的处理已按最新沟通更新;候选迁移方案尚待运行验证。
+
+[ad_ctx]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/examples/auto_deploy/model_registry/configs/disagg_ctx.yaml#L8
+[ad_gen]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/examples/auto_deploy/model_registry/configs/disagg_gen.yaml#L8
+[ad_test]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_ad_disagg.py#L187
+[ad_serve]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py#L81
+[ad_smoke]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/unittest/auto_deploy/singlegpu/smoke/test_disagg.py#L46
+[accuracy]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/accuracy/test_disaggregated_serving.py#L836
+[helix_yaml]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml#L13
+[nemotron_ucx]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml#L62
+[llama4]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml#L19
+[metrics]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml#L16
+[single_gpu]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py#L1008
+[binding_tests]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/unittest/others/test_kv_cache_transceiver.py#L418
+[cancel_gate]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py#L542
+[perf_collector]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/perf/test_perf_sanity.py#L4402
diff --git a/pr-17495-transceiver.html b/pr-17495-transceiver.html
new file mode 100644
index 000000000000..2cdc6bde507d
--- /dev/null
+++ b/pr-17495-transceiver.html
@@ -0,0 +1,184 @@
+
+
+
+
PR #17495 · Disagg 兼容设置与后续讨论
+
+
+跳至正文
+
+ENGINEERING NOTE / 2026.09.15
+PR #17495 / TRANSCEIVER
Disagg 兼容设置
与后续讨论
KVCM V2 默认启用后的 disagg 配置 review。按文件列出当前保留项、覆盖依据及待讨论方案。
CI 通过后合入,当前必要的 WAR 保留。 已有等价 V2 coverage 的重复 V1 测试已在本 PR 删除。
+当前 PR
保留必要 WAR以 CI 通过为合入前提
AutoDeploy
V1 + CPP 已确认无额外修复或迁移事项
后续讨论
测试取舍与 NIXL 方案Nemotron · 专属接口 · 指标
+Transport NIXL-based,包含 NIXL+CPPManager KVCM V1 / V2Runtime CPP / Python
+
+
+13 个文件 · 4 个分类待讨论项默认展开 · 点击文件名展开详情
+没有匹配的文件
调整关键词,或清除当前筛选。
+01AutoDeploy
保留 V1 + CPP · 无额外 fix
5 个文件AutoDeploy 继续使用 V1 manager 和 CPP runtime。本 PR 保留显式 CPP/V1 WAR,使默认切到 V2 后仍走现有执行路径,并由本 PR 的 CI 验证。这部分无需修复或迁移。
+examples/auto_deploy/model_registry/configs/disagg_ctx.yaml保留 WAR
+当前处理
backend: DEFAULT + transceiver_runtime: CPP;manager 由 AutoDeploy factory 创建为 V1。
+
+
+examples/auto_deploy/model_registry/configs/disagg_gen.yaml保留 WAR
+当前处理
backend: DEFAULT + transceiver_runtime: CPP;manager 由 AutoDeploy factory 创建为 V1。
+
+
+tests/integration/defs/disaggregated/test_ad_disagg.py保留 WAR
+
+
+tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py保留 WAR
+
+
+tests/unittest/auto_deploy/singlegpu/smoke/test_disagg.py保留 WAR
+
+Helix 的实现针对 Kimi K3 的具体需求。下列 DeepSeek/Qwen case 在本 PR 保留 CPP/V1 WAR;其他模型的支持扩展由后续需求决定。
+tests/integration/defs/accuracy/test_disaggregated_serving.py按需求保留
+当前处理
DeepSeek/Qwen Helix 的 V1+CPP;launcher 通过环境变量选择 UCX
待讨论项
若出现对应模型的支持需求,以同拓扑的精度及 overlap 测试评估 NIXL/V2 方案。
+
+
+tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml按需求保留
+当前处理
为既有 UCX+CPP 补显式 V1;保留 CTX TP2 → GEN TP1/CP2
待讨论项
NIXL/V2 支持按后续需求评估,当前保留原有拓扑覆盖。
+
+
+tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml待讨论
+当前处理
为既有 UCX+CPP 补显式 V1;保留 hybrid 模型性能场景
待讨论项
删除或迁移到 NIXL,取决于这条 Hopper workload 的维护需求与覆盖价值。若保留,再确定 NIXL 下的 manager/runtime 组合。
+
Nemotron 这份配置仍由 perf 测试自动收集。目前未找到同 H200、TP2、8k/1k、concurrency 64 的等价 NIXL 配置,删除或迁移尚未确定。本 PR 保留现有 WAR。
+
迁移方案的覆盖比较以原模型、拓扑、workload 和断言为基准。fifo_v2 表示通信实现;KVCM V2 的覆盖取决于 manager 配置和实际运行路径。
+03接口与指标
NIXL 方案待评估 · 原回归覆盖保留
3 个文件
+tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py待讨论
+当前依赖 / 限制
arbitrary-transfer 测试使用 C++ serialized DataTransceiverState;Python get_context_state() 尚未实现。本 PR 为既有 CPP case 补 V1。
待讨论项
NIXL+CPP 是候选保留方案。若已有 Python 等价接口计划,对应覆盖包括成功传输和缺块错误。
+
+
+tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml待讨论
+当前依赖 / 限制
本 PR 显式指定 CPP/V1,保留 128k input 与 2048-token C++ transfer buffer 的溢出回归。
待讨论项
NIXL+CPP 是候选迁移方案,覆盖比较关注原 buffer 路径及溢出触发条件。
+
+
+tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml待讨论
+当前依赖 / 限制
本 PR 使用 runtime/manager auto;调用方 UCX 环境变量使其选择 CPP/V1。现有 timing metrics 与 send/recv CSV 缺少等价 Python 验证。
待讨论项
待确定保留现有指标所需的 NIXL runtime。迁移涉及调用方和配置;Python 方案的覆盖以现有指标及 CSV 断言为基准。
+
上述候选方案尚未验证。覆盖是否等价取决于原断言是否通过,以及接口、buffer 和指标路径是否仍实际执行。
+以下测试验证 C++ 本身的行为,本 PR 通过显式 CPP 指定固定测试对象。这部分保留为专项回归,无 Python 功能补齐事项。
+tests/unittest/others/test_kv_cache_transceiver.py专项回归
+保留原因
五处 CPP 指定分别覆盖 C++ cancel 状态、shared_ptr/GC 生命周期、timeout warning 去重、warning 开关和 bounded polling;无 legacy selector 时使用 NIXL+CPP。
+
+
+tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py专项回归
+保留原因
四处 CPP 指定固定 backend/配置选择行为,包含 NIXL+LIBFABRIC 和 legacy selector 优先级。legacy selector 测试随对应配置接口退役再清理。
+
+
+TENSORRT-LLM · TRANSCEIVER REVIEWPR #17495 / 13 FILES
+
+
+
diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml
index a4f0f896e6a2..7d5a9abadf00 100644
--- a/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml
+++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml
@@ -1,3 +1,7 @@
+# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+
+# Pin v1 KVCM and the C++ transceiver for v1 timing metrics and CSV output.
hostname: localhost
model: TinyLlama/TinyLlama-1.1B-Chat-v1.0
free_gpu_memory_fraction: 0.25
@@ -13,10 +17,10 @@ context_servers:
return_perf_metrics: True
perf_metrics_max_requests: 1000
kv_cache_config:
- use_kv_cache_manager_v2: auto
+ use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: DEFAULT
- transceiver_runtime: auto
+ transceiver_runtime: CPP
generation_servers:
num_instances: 1
tensor_parallel_size: 1
@@ -24,7 +28,7 @@ generation_servers:
return_perf_metrics: True
perf_metrics_max_requests: 1000
kv_cache_config:
- use_kv_cache_manager_v2: auto
+ use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: DEFAULT
- transceiver_runtime: auto
+ transceiver_runtime: CPP
diff --git a/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py b/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py
index d1e9b98db876..99279bbdd534 100644
--- a/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py
+++ b/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py
@@ -599,9 +599,7 @@ def test_disaggregated_spec_dec_batch_slot_limit(model, spec_dec_model_path,
kv_cache_configs = [
KvCacheConfig(max_tokens=128,
enable_block_reuse=False,
- free_gpu_memory_fraction=0.4,
- use_kv_cache_manager_v2=eagle3_one_model)
- for _ in range(2)
+ free_gpu_memory_fraction=0.4) for _ in range(2)
]
cache_transceiver_configs = [
CacheTransceiverConfig(backend="DEFAULT") for _ in range(2)
From fef8957633424e4d95e08c41da77f6d158aaf30d Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Tue, 15 Sep 2026 04:22:00 -0700
Subject: [PATCH 09/16] [None][chore] Remove local disagg review documents
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
pr-17495-disagg-cpp-by-file.md | 93 -----------------
pr-17495-transceiver.html | 184 ---------------------------------
2 files changed, 277 deletions(-)
delete mode 100644 pr-17495-disagg-cpp-by-file.md
delete mode 100644 pr-17495-transceiver.html
diff --git a/pr-17495-disagg-cpp-by-file.md b/pr-17495-disagg-cpp-by-file.md
deleted file mode 100644
index aaaf653b9d3d..000000000000
--- a/pr-17495-disagg-cpp-by-file.md
+++ /dev/null
@@ -1,93 +0,0 @@
-
-
-# PR #17495 — Disagg 兼容设置与后续讨论
-
-[PR #17495](https://github.com/NVIDIA/TensorRT-LLM/pull/17495) 将 KVCM V2 设为默认。为保留现有 disagg 功能和回归覆盖,PR 对部分文件显式指定了 CPP runtime 或 KVCM V1。
-
-本 PR 以 CI 通过为合入前提,保留当前必要的 WAR(兼容设置),包括 AutoDeploy 的 CPP/V1 指定。AutoDeploy 已确认继续使用 V1,无需额外 fix。
-
-本文按文件列出当前处理和待讨论项,供 transceiver team review。第 2–3 节涉及的迁移和功能扩展取决于实际需求与已有覆盖,放在后续 PR 处理。已有等价 V2 coverage 的重复 V1 测试已在本 PR 删除,不列入后续跟进项。
-
-配置与支持范围:
-
-- **Transport:**团队支持 NIXL-based 方案,NIXL+CPP 在这一范围内。直接 UCX/MPI backend、NIXL 使用的 UCX plugin 和 MPI 启动设施属于不同层面的配置。
-- **Manager 与 runtime:**下文 V1/V2 指 KVCM 版本,CPP/Python 指 transceiver runtime。
-- **Helix:**目前按具体需求提供支持,已有实现对应 Kimi K3。是否扩展到 DeepSeek/Qwen 等模型取决于后续需求。
-
-## 1. AutoDeploy:保留 V1 及本 PR 的 CI WAR
-
-AutoDeploy 继续使用 V1 manager 和 CPP runtime。本 PR 保留显式 CPP/V1 WAR,使默认切到 V2 后仍走现有执行路径,并由本 PR 的 CI 验证。这部分无需修复或迁移。
-
-| 文件 | 本 PR 保留的 CI WAR |
-|---|---|
-| [examples/auto_deploy/model_registry/configs/disagg_ctx.yaml][ad_ctx] | `backend: DEFAULT` + `transceiver_runtime: CPP`;manager 由 AutoDeploy factory 创建为 V1。 |
-| [examples/auto_deploy/model_registry/configs/disagg_gen.yaml][ad_gen] | `backend: DEFAULT` + `transceiver_runtime: CPP`;manager 由 AutoDeploy factory 创建为 V1。 |
-| [tests/integration/defs/disaggregated/test_ad_disagg.py][ad_test] | 显式 V1+CPP。 |
-| [tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py][ad_serve] | 显式 V1+CPP。 |
-| [tests/unittest/auto_deploy/singlegpu/smoke/test_disagg.py][ad_smoke] | 显式 V1+CPP。 |
-
-## 2. Helix 与 Nemotron:支持范围及测试取舍
-
-Helix 的实现针对 Kimi K3 的具体需求。下列 DeepSeek/Qwen case 在本 PR 保留 CPP/V1 WAR;其他模型的支持扩展由后续需求决定。
-
-| 文件 | 本 PR 保留的设置与覆盖 | 后续讨论点 |
-|---|---|---|
-| [tests/integration/defs/accuracy/test_disaggregated_serving.py][accuracy] | DeepSeek/Qwen Helix 的 V1+CPP;launcher 通过环境变量选择 UCX | 若出现对应模型的支持需求,以同拓扑的精度及 overlap 测试评估 NIXL/V2 方案。 |
-| [tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml][helix_yaml] | 为既有 UCX+CPP 补显式 V1;保留 CTX TP2 → GEN TP1/CP2 | NIXL/V2 支持按后续需求评估,当前保留原有拓扑覆盖。 |
-| [tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml][nemotron_ucx] | 为既有 UCX+CPP 补显式 V1;保留 hybrid 模型性能场景 | 删除或迁移到 NIXL,取决于这条 Hopper workload 的维护需求与覆盖价值。若保留,再确定 NIXL 下的 manager/runtime 组合。 |
-
-Nemotron 这份配置仍由 [perf 测试自动收集][perf_collector]。目前未找到同 H200、TP2、8k/1k、concurrency 64 的等价 NIXL 配置,删除或迁移尚未确定。本 PR 保留现有 WAR。
-
-迁移方案的覆盖比较以原模型、拓扑、workload 和断言为基准。`fifo_v2` 表示通信实现;KVCM V2 的覆盖取决于 manager 配置和实际运行路径。
-
-## 3. C++ 专属接口与指标:NIXL 下的可选方案
-
-| 文件 | 当前依赖 / 限制 | 后续讨论点 |
-|---|---|---|
-| [tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py][single_gpu] | arbitrary-transfer 测试使用 C++ serialized `DataTransceiverState`;Python `get_context_state()` 尚未实现。本 PR 为既有 CPP case 补 V1。 | NIXL+CPP 是候选保留方案。若已有 Python 等价接口计划,对应覆盖包括成功传输和缺块错误。 |
-| [tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml][llama4] | 本 PR 显式指定 CPP/V1,保留 128k input 与 2048-token C++ transfer buffer 的溢出回归。 | NIXL+CPP 是候选迁移方案,覆盖比较关注原 buffer 路径及溢出触发条件。 |
-| [tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml][metrics] | 本 PR 使用 runtime/manager `auto`;调用方 UCX 环境变量使其选择 CPP/V1。现有 timing metrics 与 send/recv CSV 缺少等价 Python 验证。 | 待确定保留现有指标所需的 NIXL runtime。迁移涉及调用方和配置;Python 方案的覆盖以现有指标及 CSV 断言为基准。 |
-
-上述候选方案尚未验证。覆盖是否等价取决于原断言是否通过,以及接口、buffer 和指标路径是否仍实际执行。
-
-## 4. 保留的 C++ 专项回归
-
-以下测试验证 C++ 本身的行为,本 PR 通过显式 CPP 指定固定测试对象。这部分保留为专项回归,无 Python 功能补齐事项。
-
-| 文件 | 保留原因 |
-|---|---|
-| [tests/unittest/others/test_kv_cache_transceiver.py][binding_tests] | 五处 CPP 指定分别覆盖 C++ cancel 状态、shared_ptr/GC 生命周期、timeout warning 去重、warning 开关和 bounded polling;无 legacy selector 时使用 NIXL+CPP。 |
-| [tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py][cancel_gate] | 四处 CPP 指定固定 backend/配置选择行为,包含 NIXL+LIBFABRIC 和 legacy selector 优先级。legacy selector 测试随对应配置接口退役再清理。 |
-
-当前待讨论项集中在 Nemotron 的覆盖取舍,以及第 3 节的 NIXL 接口与指标方案。后续若确认等价覆盖,可据此评估撤回对应 WAR,并用 CI 验证。本 PR 保留当前必要的 WAR,以 CI 通过后合入为目标。
-
-代码核查版本:`5eed11ca57`,下方文件链接固定到该提交。AutoDeploy 和 Helix 的处理已按最新沟通更新;候选迁移方案尚待运行验证。
-
-[ad_ctx]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/examples/auto_deploy/model_registry/configs/disagg_ctx.yaml#L8
-[ad_gen]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/examples/auto_deploy/model_registry/configs/disagg_gen.yaml#L8
-[ad_test]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_ad_disagg.py#L187
-[ad_serve]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py#L81
-[ad_smoke]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/unittest/auto_deploy/singlegpu/smoke/test_disagg.py#L46
-[accuracy]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/accuracy/test_disaggregated_serving.py#L836
-[helix_yaml]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml#L13
-[nemotron_ucx]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml#L62
-[llama4]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml#L19
-[metrics]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml#L16
-[single_gpu]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py#L1008
-[binding_tests]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/unittest/others/test_kv_cache_transceiver.py#L418
-[cancel_gate]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py#L542
-[perf_collector]: https://github.com/yizhang-nv/TensorRT-LLM/blob/5eed11ca57b4224a3896d6ab30209305ea1c7e4e/tests/integration/defs/perf/test_perf_sanity.py#L4402
diff --git a/pr-17495-transceiver.html b/pr-17495-transceiver.html
deleted file mode 100644
index 2cdc6bde507d..000000000000
--- a/pr-17495-transceiver.html
+++ /dev/null
@@ -1,184 +0,0 @@
-
-
-
-PR #17495 · Disagg 兼容设置与后续讨论
-
-
-跳至正文
-
-ENGINEERING NOTE / 2026.09.15
-PR #17495 / TRANSCEIVER
Disagg 兼容设置
与后续讨论
KVCM V2 默认启用后的 disagg 配置 review。按文件列出当前保留项、覆盖依据及待讨论方案。
CI 通过后合入,当前必要的 WAR 保留。 已有等价 V2 coverage 的重复 V1 测试已在本 PR 删除。
-当前 PR
保留必要 WAR以 CI 通过为合入前提
AutoDeploy
V1 + CPP 已确认无额外修复或迁移事项
后续讨论
测试取舍与 NIXL 方案Nemotron · 专属接口 · 指标
-Transport NIXL-based,包含 NIXL+CPPManager KVCM V1 / V2Runtime CPP / Python
-
-
-13 个文件 · 4 个分类待讨论项默认展开 · 点击文件名展开详情
-没有匹配的文件
调整关键词,或清除当前筛选。
-01AutoDeploy
保留 V1 + CPP · 无额外 fix
5 个文件AutoDeploy 继续使用 V1 manager 和 CPP runtime。本 PR 保留显式 CPP/V1 WAR,使默认切到 V2 后仍走现有执行路径,并由本 PR 的 CI 验证。这部分无需修复或迁移。
-examples/auto_deploy/model_registry/configs/disagg_ctx.yaml保留 WAR
-当前处理
backend: DEFAULT + transceiver_runtime: CPP;manager 由 AutoDeploy factory 创建为 V1。
-
-
-examples/auto_deploy/model_registry/configs/disagg_gen.yaml保留 WAR
-当前处理
backend: DEFAULT + transceiver_runtime: CPP;manager 由 AutoDeploy factory 创建为 V1。
-
-
-tests/integration/defs/disaggregated/test_ad_disagg.py保留 WAR
-
-
-tests/integration/defs/disaggregated/test_ad_disagg_trtllm_serve.py保留 WAR
-
-
-tests/unittest/auto_deploy/singlegpu/smoke/test_disagg.py保留 WAR
-
-Helix 的实现针对 Kimi K3 的具体需求。下列 DeepSeek/Qwen case 在本 PR 保留 CPP/V1 WAR;其他模型的支持扩展由后续需求决定。
-tests/integration/defs/accuracy/test_disaggregated_serving.py按需求保留
-当前处理
DeepSeek/Qwen Helix 的 V1+CPP;launcher 通过环境变量选择 UCX
待讨论项
若出现对应模型的支持需求,以同拓扑的精度及 overlap 测试评估 NIXL/V2 方案。
-
-
-tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml按需求保留
-当前处理
为既有 UCX+CPP 补显式 V1;保留 CTX TP2 → GEN TP1/CP2
待讨论项
NIXL/V2 支持按后续需求评估,当前保留原有拓扑覆盖。
-
-
-tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml待讨论
-当前处理
为既有 UCX+CPP 补显式 V1;保留 hybrid 模型性能场景
待讨论项
删除或迁移到 NIXL,取决于这条 Hopper workload 的维护需求与覆盖价值。若保留,再确定 NIXL 下的 manager/runtime 组合。
-
Nemotron 这份配置仍由 perf 测试自动收集。目前未找到同 H200、TP2、8k/1k、concurrency 64 的等价 NIXL 配置,删除或迁移尚未确定。本 PR 保留现有 WAR。
-
迁移方案的覆盖比较以原模型、拓扑、workload 和断言为基准。fifo_v2 表示通信实现;KVCM V2 的覆盖取决于 manager 配置和实际运行路径。
-03接口与指标
NIXL 方案待评估 · 原回归覆盖保留
3 个文件
-tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py待讨论
-当前依赖 / 限制
arbitrary-transfer 测试使用 C++ serialized DataTransceiverState;Python get_context_state() 尚未实现。本 PR 为既有 CPP case 补 V1。
待讨论项
NIXL+CPP 是候选保留方案。若已有 Python 等价接口计划,对应覆盖包括成功传输和缺块错误。
-
-
-tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml待讨论
-当前依赖 / 限制
本 PR 显式指定 CPP/V1,保留 128k input 与 2048-token C++ transfer buffer 的溢出回归。
待讨论项
NIXL+CPP 是候选迁移方案,覆盖比较关注原 buffer 路径及溢出触发条件。
-
-
-tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml待讨论
-当前依赖 / 限制
本 PR 使用 runtime/manager auto;调用方 UCX 环境变量使其选择 CPP/V1。现有 timing metrics 与 send/recv CSV 缺少等价 Python 验证。
待讨论项
待确定保留现有指标所需的 NIXL runtime。迁移涉及调用方和配置;Python 方案的覆盖以现有指标及 CSV 断言为基准。
-
上述候选方案尚未验证。覆盖是否等价取决于原断言是否通过,以及接口、buffer 和指标路径是否仍实际执行。
-以下测试验证 C++ 本身的行为,本 PR 通过显式 CPP 指定固定测试对象。这部分保留为专项回归,无 Python 功能补齐事项。
-tests/unittest/others/test_kv_cache_transceiver.py专项回归
-保留原因
五处 CPP 指定分别覆盖 C++ cancel 状态、shared_ptr/GC 生命周期、timeout warning 去重、warning 开关和 bounded polling;无 legacy selector 时使用 NIXL+CPP。
-
-
-tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py专项回归
-保留原因
四处 CPP 指定固定 backend/配置选择行为,包含 NIXL+LIBFABRIC 和 legacy selector 优先级。legacy selector 测试随对应配置接口退役再清理。
-
-
-TENSORRT-LLM · TRANSCEIVER REVIEWPR #17495 / 13 FILES
-
-
-
From 3abf01ab0608d2aa199601b99b26e3dc9a41869a Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Wed, 16 Sep 2026 01:31:56 -0700
Subject: [PATCH 10/16] [None][test] Waive single-GPU failures pending KVCM V2
fixes
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/test_lists/waives.txt | 18 ++++++++++++++++++
1 file changed, 18 insertions(+)
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index 3c4a86dd6600..3b933fd95b01 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -85,6 +85,11 @@ examples/visual_gen/test_visual_gen_wan.py::test_wan_feature_accuracy_against_go
examples/visual_gen/test_visual_gen_wan.py::test_wan_feature_accuracy_against_golden[wan22-cuda-graph] SKIP (https://nvbugs/6572800)
examples/visual_gen/test_visual_gen_wan.py::test_wan_feature_accuracy_against_golden[wan22-fp8-blockwise] SKIP (https://nvbugs/6572800)
examples/visual_gen/test_visual_gen_wan.py::test_wan_feature_accuracy_against_golden[wan22-nvfp4] SKIP (https://nvbugs/6572800)
+full:A10/test_e2e.py::test_openai_chat_multimodal_example SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
+full:A10/test_e2e.py::test_trtllm_serve_multimodal_example SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
+full:A10/unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_image SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
+full:A10/unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_image_streaming SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
+full:A10/unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_rgba_image SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
full:A100/accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_nvfp4_4gpus_hopper_w4a16 SKIP (https://nvbugs/6802472)
full:A100/accuracy/test_llm_api_pytorch_multimodal.py::TestExaone4_5_33B::test_auto_dtype[forced_chunked_prefill] SKIP (https://nvbugs/6597570)
full:A100/accuracy/test_llm_api_pytorch_multimodal.py::TestExaone4_5_33B::test_auto_dtype[full_budget] SKIP (https://nvbugs/6597570)
@@ -118,10 +123,23 @@ full:B300/disaggregated/test_disaggregated.py::test_disaggregated_mamba_conc_gre
full:B300/disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp4-TinyLlama-1.1B-Chat-v1.0] SKIP (https://nvbugs/6728119)
full:B300/disaggregated/test_disaggregated.py::test_disaggregated_qwen3_32b_fp8[Qwen3/Qwen3-32B-FP8] SKIP (https://nvbugs/6770977)
full:B300/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke SKIP (https://nvbugs/6782589)
+full:B300/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch SKIP (KVCM V2 LoRA logprob mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19206)
full:DGX_B200/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_bfloat16[mtp_nextn=2-attention_dp=True-cuda_graph=True-overlap_scheduler=True-torch_compile=False-enable_chunked_prefill=False-v2_kv_cache=True] SKIP (https://nvbugs/6633268)
+full:DGX_B200/accuracy/test_llm_api_pytorch.py::TestSeedOss_36B::test_auto_dtype SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
+full:DGX_B200/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch SKIP (KVCM V2 LoRA logprob mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19206)
full:DGX_B200/unittest/tools/test_layer_wise_benchmarks.py::test_performance_alignment[1] SKIP (https://nvbugs/6669275)
full:DGX_H100/accuracy/test_disaggregated_serving.py::TestQwen3_5_4B::test_mismatched_block_reuse SKIP (https://nvbugs/6793949)
full:DGX_H100/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales[mtp=disable-fp8kv=True-attention_dp=False-cuda_graph=True-overlap_scheduler=True-torch_compile=True] SKIP (https://nvbugs/6700265)
+full:DGX_H100/accuracy/test_llm_api_pytorch_multimodal.py::TestMistralSmall24B::test_auto_dtype[forced_chunked_prefill] SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
+full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[guaranteed-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
+full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[max-util-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
+full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[no-overlap-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
+full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[offload-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
+full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[offload-no-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
+full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[python-scheduler] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
+full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
+full:DGX_H100/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch SKIP (KVCM V2 LoRA logprob mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19206)
+full:DGX_H100/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py::test_correctness_across_batch_sizes[model_drafter-schedule1] SKIP (KVCM V2 dynamic draft warmup shape mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19204)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy SKIP (https://nvbugs/6276923)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_contention_opt SKIP (https://nvbugs/6276923)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_mode_b_overlap SKIP (https://nvbugs/6276923)
From 74da14b756efba0b62001b79e3902b0dc39d869a Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Wed, 16 Sep 2026 02:13:03 -0700
Subject: [PATCH 11/16] [None][test] Remove stale V1 perf overrides and
redundant test
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
...tx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...tx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...tx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...tx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml | 2 --
...x1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml | 2 --
...tx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml | 2 --
...1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml | 2 --
.../kv_cache/test_kv_cache_estimation.py | 30 -------------------
16 files changed, 60 deletions(-)
diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
index d5046594c038..384f198ca17e 100644
--- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
@@ -57,7 +57,6 @@ worker_config:
enable_padding: true
max_batch_size: 1536
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -81,7 +80,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
index ebc2bac0df08..f18b38e6dc45 100644
--- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
@@ -57,7 +57,6 @@ worker_config:
enable_padding: true
max_batch_size: 1536
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -81,7 +80,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
index 8cafded2577a..685171bf899e 100644
--- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
@@ -53,7 +53,6 @@ worker_config:
enable_padding: true
max_batch_size: 256
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -78,7 +77,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
index 1f6ad52274b9..97ca603485ce 100644
--- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
@@ -65,7 +65,6 @@ worker_config:
enable_padding: true
max_batch_size: 1280
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.85
dtype: fp8
@@ -94,7 +93,6 @@ worker_config:
enable_padding: true
max_batch_size: 30
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.8
dtype: fp8
diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
index 45f8e986fd98..c031ad48ecbd 100644
--- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
@@ -53,7 +53,6 @@ worker_config:
enable_padding: true
max_batch_size: 1024
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -78,7 +77,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
index c2eb718be285..11b64deb265e 100644
--- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
@@ -53,7 +53,6 @@ worker_config:
enable_padding: true
max_batch_size: 1024
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -78,7 +77,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
index 27aecee14739..b6bde7daf105 100644
--- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
@@ -57,7 +57,6 @@ worker_config:
enable_padding: true
max_batch_size: 512
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -81,7 +80,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_llama-3.1-8b-bf16_1k1k_con256_ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_llama-3.1-8b-bf16_1k1k_con256_ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml
index cc3db00acce4..38cbadd729c7 100644
--- a/tests/scripts/perf-sanity/disaggregated/gb200_llama-3.1-8b-bf16_1k1k_con256_ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf-sanity/disaggregated/gb200_llama-3.1-8b-bf16_1k1k_con256_ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml
@@ -68,7 +68,6 @@ worker_config:
enable_padding: true
max_batch_size: 256
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.85
dtype: auto
@@ -90,7 +89,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.85
dtype: auto
diff --git a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
index ffa8cdfa4522..94f9074fb86d 100644
--- a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
@@ -65,7 +65,6 @@ worker_config:
enable_padding: true
max_batch_size: 1280
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.85
dtype: fp8
@@ -94,7 +93,6 @@ worker_config:
enable_padding: true
max_batch_size: 30
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.8
dtype: fp8
diff --git a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
index 1d16d05a1ad3..b402a1ddaeb4 100644
--- a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
@@ -64,7 +64,6 @@ worker_config:
enable_padding: true
max_batch_size: 1024
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -89,7 +88,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
index 9604ca2743f9..d29d122474e7 100644
--- a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml
@@ -64,7 +64,6 @@ worker_config:
enable_padding: true
max_batch_size: 1024
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -89,7 +88,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
index 6740b7f581ec..005fef248f59 100644
--- a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml
@@ -68,7 +68,6 @@ worker_config:
enable_padding: true
max_batch_size: 512
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -92,7 +91,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf/disaggregated/gb200_stress-gpt-oss-120b-fp4_8k1k_ctx1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_stress-gpt-oss-120b-fp4_8k1k_ctx1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml
index f7c14976709b..7b97ad0e0589 100644
--- a/tests/scripts/perf/disaggregated/gb200_stress-gpt-oss-120b-fp4_8k1k_ctx1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb200_stress-gpt-oss-120b-fp4_8k1k_ctx1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml
@@ -70,7 +70,6 @@ worker_config:
enable_padding: true
max_batch_size: 1024
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
@@ -99,7 +98,6 @@ worker_config:
enable_attention_dp: false
cuda_graph_config: null
kv_cache_config:
- use_kv_cache_manager_v2: false
enable_block_reuse: false
free_gpu_memory_fraction: 0.9
dtype: fp8
diff --git a/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_128k8k_con256_ctx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_128k8k_con256_ctx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml
index d109498ebdbe..c3ac0ccb63a2 100644
--- a/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_128k8k_con256_ctx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_128k8k_con256_ctx1_pp4_gen1_dep8_eplb0_mtp1_ccb-NIXL.yaml
@@ -67,7 +67,6 @@ worker_config:
max_batch_size: 16
kv_cache_config:
enable_block_reuse: false
- use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.8
dtype: fp8
moe_config:
@@ -96,7 +95,6 @@ worker_config:
cuda_graph_config: null
kv_cache_config:
enable_block_reuse: false
- use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.3
dtype: fp8
moe_config:
diff --git a/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml
index e67c1a90ca82..02ac4618d977 100644
--- a/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml
+++ b/tests/scripts/perf/disaggregated/gb300_deepseek-r1-fp4_8k1k_con4096_ctx1_dep4_gen1_dep16_eplb0_mtp1_ccb-NIXL.yaml
@@ -66,7 +66,6 @@ worker_config:
max_batch_size: 256
kv_cache_config:
enable_block_reuse: false
- use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.8
dtype: fp8
moe_config:
@@ -95,7 +94,6 @@ worker_config:
cuda_graph_config: null
kv_cache_config:
enable_block_reuse: false
- use_kv_cache_manager_v2: false
free_gpu_memory_fraction: 0.6
dtype: fp8
moe_config:
diff --git a/tests/unittest/_torch/executor/kv_cache/test_kv_cache_estimation.py b/tests/unittest/_torch/executor/kv_cache/test_kv_cache_estimation.py
index 452a5e27b6b1..2d698874e9a8 100644
--- a/tests/unittest/_torch/executor/kv_cache/test_kv_cache_estimation.py
+++ b/tests/unittest/_torch/executor/kv_cache/test_kv_cache_estimation.py
@@ -1255,36 +1255,6 @@ def test_estimation_temporarily_uses_inferred_pool_sizing(
assert kv_cache_config.avg_seq_len == avg_seq_len
-@pytest.mark.parametrize(
- ("is_v2", "cp_config", "expected_skip_est"),
- [
- (True, {}, True),
- (False, {}, False),
- (True, {"cp_type": "ring"}, False),
- ],
-)
-def test_vanilla_attention_uses_capacity_fallback_for_v2(
- is_v2: bool,
- cp_config: dict,
- expected_skip_est: bool,
-) -> None:
- creator = object.__new__(KvCacheCreator)
- creator._skip_est = False
- creator._mapping = SimpleNamespace(cp_config=cp_config)
- creator._model_engine = SimpleNamespace(
- model=SimpleNamespace(
- model_config=SimpleNamespace(
- attn_backend="VANILLA",
- is_encoder_decoder=False,
- )
- )
- )
- creator._is_kv_cache_manager_v2 = is_v2
-
- assert creator.try_prepare_estimation() is False
- assert creator._skip_est is expected_skip_est
-
-
@pytest.mark.parametrize(
("estimating_kv_cache", "expected_avg_seq_len"),
[(True, 2045), (False, 2055)],
From 172bc6bfbab1c9aaa70f18c17a66d250859ba166 Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Wed, 16 Sep 2026 08:25:07 -0700
Subject: [PATCH 12/16] [None][test] Re-enable Qwen3 LoRA after merged V2 test
fix
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/test_lists/waives.txt | 3 ---
1 file changed, 3 deletions(-)
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index 3b933fd95b01..637932d00207 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -123,10 +123,8 @@ full:B300/disaggregated/test_disaggregated.py::test_disaggregated_mamba_conc_gre
full:B300/disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_first[ctx_pp4-TinyLlama-1.1B-Chat-v1.0] SKIP (https://nvbugs/6728119)
full:B300/disaggregated/test_disaggregated.py::test_disaggregated_qwen3_32b_fp8[Qwen3/Qwen3-32B-FP8] SKIP (https://nvbugs/6770977)
full:B300/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke SKIP (https://nvbugs/6782589)
-full:B300/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch SKIP (KVCM V2 LoRA logprob mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19206)
full:DGX_B200/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_bfloat16[mtp_nextn=2-attention_dp=True-cuda_graph=True-overlap_scheduler=True-torch_compile=False-enable_chunked_prefill=False-v2_kv_cache=True] SKIP (https://nvbugs/6633268)
full:DGX_B200/accuracy/test_llm_api_pytorch.py::TestSeedOss_36B::test_auto_dtype SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
-full:DGX_B200/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch SKIP (KVCM V2 LoRA logprob mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19206)
full:DGX_B200/unittest/tools/test_layer_wise_benchmarks.py::test_performance_alignment[1] SKIP (https://nvbugs/6669275)
full:DGX_H100/accuracy/test_disaggregated_serving.py::TestQwen3_5_4B::test_mismatched_block_reuse SKIP (https://nvbugs/6793949)
full:DGX_H100/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales[mtp=disable-fp8kv=True-attention_dp=False-cuda_graph=True-overlap_scheduler=True-torch_compile=True] SKIP (https://nvbugs/6700265)
@@ -138,7 +136,6 @@ full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareSche
full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[offload-no-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[python-scheduler] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
-full:DGX_H100/unittest/_torch/modules/tests_lora_modules/test_qwen3_sanity.py::TestQwen3LoRA::test_qwen3_bf16_lora_cuda_graph_specialization_mixed_batch SKIP (KVCM V2 LoRA logprob mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19206)
full:DGX_H100/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py::test_correctness_across_batch_sizes[model_drafter-schedule1] SKIP (KVCM V2 dynamic draft warmup shape mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19204)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy SKIP (https://nvbugs/6276923)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_contention_opt SKIP (https://nvbugs/6276923)
From 3a18425189dbbdeea872ca7dd41bcc972169839d Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Thu, 17 Sep 2026 05:48:51 -0700
Subject: [PATCH 13/16] [None][test] Re-enable dynamic draft after merged
warmup fix
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/test_lists/waives.txt | 1 -
1 file changed, 1 deletion(-)
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index 637932d00207..d3d5ec23614b 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -136,7 +136,6 @@ full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareSche
full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[offload-no-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[python-scheduler] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
-full:DGX_H100/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py::test_correctness_across_batch_sizes[model_drafter-schedule1] SKIP (KVCM V2 dynamic draft warmup shape mismatch; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19204)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy SKIP (https://nvbugs/6276923)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_contention_opt SKIP (https://nvbugs/6276923)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_mode_b_overlap SKIP (https://nvbugs/6276923)
From 5905384ae85c1009b258effc31cc33454ca1be52 Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Fri, 18 Sep 2026 00:43:49 -0700
Subject: [PATCH 14/16] [None][test] Update KVCM V2 CI coverage and waivers
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/defs/examples/test_ray.py | 6 ------
tests/integration/test_lists/test-db/l0_dgx_b200.yml | 1 -
tests/integration/test_lists/test-db/l0_dgx_h100.yml | 1 -
tests/integration/test_lists/waives.txt | 2 ++
4 files changed, 2 insertions(+), 8 deletions(-)
diff --git a/tests/integration/defs/examples/test_ray.py b/tests/integration/defs/examples/test_ray.py
index 201192d5fdd1..5023fa8d99de 100644
--- a/tests/integration/defs/examples/test_ray.py
+++ b/tests/integration/defs/examples/test_ray.py
@@ -70,12 +70,6 @@ def test_llm_inference_distributed_ray(ray_example_root, llm_venv, tp_size,
venv_check_call(llm_venv, cmd)
-@pytest.mark.skip_less_device(2)
-@pytest.mark.parametrize("tp_size", [1, 2], ids=["tp1", "tp2"])
-def test_ray_disaggregated_serving(ray_example_root, llm_venv, tp_size):
- _run_ray_disaggregated_serving(ray_example_root, tp_size, "NIXL", "CPP")
-
-
@pytest.mark.skip_less_device(2)
@pytest.mark.parametrize("tp_size", [1, 2], ids=["tp1", "tp2"])
def test_ray_disaggregated_serving_python(ray_example_root, llm_venv, tp_size):
diff --git a/tests/integration/test_lists/test-db/l0_dgx_b200.yml b/tests/integration/test_lists/test-db/l0_dgx_b200.yml
index 91ac09081e6b..a194d0175b86 100644
--- a/tests/integration/test_lists/test-db/l0_dgx_b200.yml
+++ b/tests/integration/test_lists/test-db/l0_dgx_b200.yml
@@ -144,7 +144,6 @@ l0_dgx_b200:
- disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_gentp2[TinyLlama-1.1B-Chat-v1.0]
- disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_gentp4[TinyLlama-1.1B-Chat-v1.0]
- examples/test_ray.py::test_llm_inference_distributed_ray[tp2pp2]
- - examples/test_ray.py::test_ray_disaggregated_serving[tp2]
- examples/test_ray.py::test_ray_disaggregated_serving_python[tp2]
- condition:
ranges:
diff --git a/tests/integration/test_lists/test-db/l0_dgx_h100.yml b/tests/integration/test_lists/test-db/l0_dgx_h100.yml
index edb497c561e2..563a78b8d8f5 100644
--- a/tests/integration/test_lists/test-db/l0_dgx_h100.yml
+++ b/tests/integration/test_lists/test-db/l0_dgx_h100.yml
@@ -291,7 +291,6 @@ l0_dgx_h100:
- examples/test_ray.py::test_llm_inference_distributed_ray[tp2]
- examples/test_ray.py::test_llm_inference_distributed_ray[pp2]
- examples/test_ray.py::test_llm_inference_distributed_ray[tep2]
- - examples/test_ray.py::test_ray_disaggregated_serving[tp1]
- examples/test_ray.py::test_ray_disaggregated_serving_python[tp1]
- condition:
ranges:
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index d3d5ec23614b..17471c1d419c 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -310,6 +310,7 @@ unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_very_long_uuid SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (https://nvbugs/6813129)
+unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_with_block_reuse SKIP (Concurrent initialization OOM; https://nv/trt-llm-cicd/job/main/job/L0_MergeRequest_PR/60801/)
unittest/_torch/ray_orchestrator/multi_gpu/test_llm_update_weights_multi_gpu.py::test_llm_update_weights_nemotron_h SKIP (https://nvbugs/6729495)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[False] SKIP (https://nvbugs/6535767)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[True] SKIP (https://nvbugs/6535767)
@@ -331,6 +332,7 @@ unittest/llmapi/test_llm_multi_gpu_pytorch.py -m "gpu2" SKIP (https://nvbugs/642
unittest/llmapi/test_llm_multi_gpu_pytorch.py::test_llm_get_stats_pp2[False-False-True] SKIP (https://nvbugs/6432826)
unittest/llmapi/test_llm_pytorch.py::test_gqa_nemo_lora[None] SKIP (https://nvbugs/6162504)
unittest/llmapi/test_llm_pytorch.py::test_gqa_nemo_lora[cuda_graph_config0] SKIP (https://nvbugs/6162504)
+unittest/llmapi/test_llm_pytorch.py::test_llm_context_only_timed_out_kv_cache_exhausted[PYTHON-NIXL-100] SKIP (Intermittent context-only timeout; https://nv/trt-llm-cicd/job/main/job/L0_MergeRequest_PR/60801/)
unittest/llmapi/test_memory_profiling.py::test_profile_kvcache SKIP (https://nvbugs/5580781)
unittest/usage/test_llmapi_config_telemetry_docs.py::test_build_capture_manifest_matches_committed_golden SKIP (https://nvbugs/6811958)
verl/test_verl_cases.py::test_async_generate SKIP (https://nvbugs/6683838)
From 11d06ea0967a324fd9959cc6b2c28d3e67e2482c Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Sun, 20 Sep 2026 08:14:37 -0700
Subject: [PATCH 15/16] [None][test] Unwaive KVCM V2 cases except
initialization OOM
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
tests/integration/test_lists/waives.txt | 20 --------------------
1 file changed, 20 deletions(-)
diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt
index 17471c1d419c..dce7568ed227 100644
--- a/tests/integration/test_lists/waives.txt
+++ b/tests/integration/test_lists/waives.txt
@@ -129,13 +129,6 @@ full:DGX_B200/unittest/tools/test_layer_wise_benchmarks.py::test_performance_ali
full:DGX_H100/accuracy/test_disaggregated_serving.py::TestQwen3_5_4B::test_mismatched_block_reuse SKIP (https://nvbugs/6793949)
full:DGX_H100/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales[mtp=disable-fp8kv=True-attention_dp=False-cuda_graph=True-overlap_scheduler=True-torch_compile=True] SKIP (https://nvbugs/6700265)
full:DGX_H100/accuracy/test_llm_api_pytorch_multimodal.py::TestMistralSmall24B::test_auto_dtype[forced_chunked_prefill] SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213)
-full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[guaranteed-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
-full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[max-util-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
-full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[no-overlap-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
-full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[offload-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
-full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[offload-no-chunked] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
-full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix[python-scheduler] SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
-full:DGX_H100/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke SKIP (KVCM V2 prefix scheduling timeout; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19202)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy SKIP (https://nvbugs/6276923)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_contention_opt SKIP (https://nvbugs/6276923)
full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_mode_b_overlap SKIP (https://nvbugs/6276923)
@@ -297,20 +290,8 @@ unittest/_torch/modeling/test_modeling_nemotron_h_multimodal.py::test_nemotron_n
unittest/_torch/modules/tests_lora_modules/test_nemotron_h_lora_sanity.py::TestNemotronHLoRA::test_lora_pp2_sanity SKIP (https://nvbugs/6428124)
unittest/_torch/moe/test_moe_backend.py::test_moe_backend[act=Relu2-e60_k4_h2048_i1408-seq=8-dtype=torch.bfloat16-backend=TRTLLM-quant=NVFP4-routing=Renormalize] SKIP (https://nvbugs/5989912)
unittest/_torch/multi_gpu/test_linear.py::test_row_linear_norm_fusion[2-hidden:16-seqlen:2] SKIP (https://nvbugs/6501404)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_epd_disagg_mm_hash_kv_cache_reuse[prompts0] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_epd_disagg_mm_hash_kv_cache_reuse[prompts0] SKIP (https://nvbugs/6812085)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_epd_disagg_mm_hash_kv_cache_reuse[prompts1] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_partial_uuids[uuids0-expected_patterns0] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_partial_uuids[uuids1-expected_patterns1] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_partial_uuids[uuids2-expected_patterns2] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_reuse[prompts1-2] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_uuid[False-hex] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_uuid[True-uuid] SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_uuid_multiple_prompts SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_kv_event_mm_keys_with_very_long_uuid SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (KVCM V2 follow-up: https://github.com/yizhang-nv/TensorRT-LLM/commit/3b419058b4e862c14c91a908bc90122e5eda44fe)
unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_no_reuse_when_disabled SKIP (https://nvbugs/6813129)
-unittest/_torch/multimodal/test_mm_encoder_standalone.py::test_pd_disagg_multimodal_with_block_reuse SKIP (Concurrent initialization OOM; https://nv/trt-llm-cicd/job/main/job/L0_MergeRequest_PR/60801/)
unittest/_torch/ray_orchestrator/multi_gpu/test_llm_update_weights_multi_gpu.py::test_llm_update_weights_nemotron_h SKIP (https://nvbugs/6729495)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[False] SKIP (https://nvbugs/6535767)
unittest/_torch/speculative/hw_agnostic/test_dflash.py::test_dflash_qwen3_5_4b[True] SKIP (https://nvbugs/6535767)
@@ -332,7 +313,6 @@ unittest/llmapi/test_llm_multi_gpu_pytorch.py -m "gpu2" SKIP (https://nvbugs/642
unittest/llmapi/test_llm_multi_gpu_pytorch.py::test_llm_get_stats_pp2[False-False-True] SKIP (https://nvbugs/6432826)
unittest/llmapi/test_llm_pytorch.py::test_gqa_nemo_lora[None] SKIP (https://nvbugs/6162504)
unittest/llmapi/test_llm_pytorch.py::test_gqa_nemo_lora[cuda_graph_config0] SKIP (https://nvbugs/6162504)
-unittest/llmapi/test_llm_pytorch.py::test_llm_context_only_timed_out_kv_cache_exhausted[PYTHON-NIXL-100] SKIP (Intermittent context-only timeout; https://nv/trt-llm-cicd/job/main/job/L0_MergeRequest_PR/60801/)
unittest/llmapi/test_memory_profiling.py::test_profile_kvcache SKIP (https://nvbugs/5580781)
unittest/usage/test_llmapi_config_telemetry_docs.py::test_build_capture_manifest_matches_committed_golden SKIP (https://nvbugs/6811958)
verl/test_verl_cases.py::test_async_generate SKIP (https://nvbugs/6683838)
From a27e923cc5a4d757ab44fe9bbfd7a1e0016aed33 Mon Sep 17 00:00:00 2001
From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
Date: Sat, 26 Sep 2026 23:28:54 -0700
Subject: [PATCH 16/16] [None][test] Initialize max beam width in KV cache
budget fixture
Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com>
---
.../_torch/executor/kv_cache/test_kv_cache_budget_split.py | 1 +
1 file changed, 1 insertion(+)
diff --git a/tests/unittest/_torch/executor/kv_cache/test_kv_cache_budget_split.py b/tests/unittest/_torch/executor/kv_cache/test_kv_cache_budget_split.py
index de1a4ae95ddc..0048c2200372 100644
--- a/tests/unittest/_torch/executor/kv_cache/test_kv_cache_budget_split.py
+++ b/tests/unittest/_torch/executor/kv_cache/test_kv_cache_budget_split.py
@@ -64,6 +64,7 @@ def _make_creator(
c._max_seq_len = 1024
c._max_num_tokens = 0
c._max_batch_size = 1
+ c._max_beam_width = 1
c._is_disagg = False
c._cache_transceiver_config = None
c._speculative_config = None