Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -44,16 +44,14 @@ cat >${work_path}/ctx_config.yaml << EOL
disable_overlap_scheduler: True
internal_request_auth_key: ${internal_request_auth_key}
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
backend: NIXL
max_tokens_in_buffer: 2048
EOL

cat >${work_path}/gen_config.yaml << EOL
internal_request_auth_key: ${internal_request_auth_key}
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
backend: NIXL
max_tokens_in_buffer: 2048
EOL

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,5 @@
# not yet supported in disaggregated context server architectures.
disable_overlap_scheduler: True
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
backend: NIXL
max_tokens_in_buffer: 2048
Original file line number Diff line number Diff line change
@@ -1,4 +1,3 @@
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
backend: NIXL
max_tokens_in_buffer: 2048
6 changes: 2 additions & 4 deletions examples/dwdp/reproduce.py
Original file line number Diff line number Diff line change
Expand Up @@ -220,8 +220,7 @@ def build_worker_config(experiment: Dict[str, Any]) -> Dict[str, Any]:
"backend": "CUTEDSL",
},
"cache_transceiver_config": {
"backend": "UCX",
"transceiver_runtime": "CPP",
"backend": "NIXL",
"max_tokens_in_buffer": max_tokens_in_buffer,
},
"num_postprocess_workers": 4,
Expand Down Expand Up @@ -256,8 +255,7 @@ def build_worker_config(experiment: Dict[str, Any]) -> Dict[str, Any]:
"free_gpu_memory_fraction": 0.3,
},
"cache_transceiver_config": {
"backend": "UCX",
"transceiver_runtime": "CPP",
"backend": "NIXL",
"max_tokens_in_buffer": max_tokens_in_buffer,
},
"moe_config": {
Expand Down
4 changes: 2 additions & 2 deletions examples/llm-api/quickstart_advanced.py
Original file line number Diff line number Diff line change
Expand Up @@ -151,11 +151,11 @@ def add_llm_args(parser):
action='store_true')
parser.add_argument(
'--use_kv_cache_manager_v2',
default='auto',
default=True,
type=_parse_kv_cache_manager_v2,
metavar='{auto,true,false}',
help=
'Whether to use KVCacheManagerV2 for KV cache management (PyTorch backend). Defaults to model-specific auto selection.',
'Whether to use KVCacheManagerV2 for KV cache management (PyTorch backend). Defaults to true; use auto for model-specific selection.',
)

# Runtime
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@ ATTACH_MODE=false
MODEL_DIR="TinyLlama/TinyLlama-1.1B-Chat-v1.0"
TP_SIZE=1
TRANSCEIVER_BACKEND="NIXL"
TRANSCEIVER_RUNTIME="CPP"
TRANSCEIVER_RUNTIME="PYTHON"
USAGE="Usage: $0 [--executor ray|mpi] [--attach] [--model model_dir] [--tp_size N] [--transceiver_backend UCX|NIXL] [--transceiver_runtime CPP|PYTHON] [--help]"

while [[ $# -gt 0 ]]; do
Expand Down Expand Up @@ -45,7 +45,7 @@ while [[ $# -gt 0 ]]; do
echo " --model model_dir Model directory (default: TinyLlama/TinyLlama-1.1B-Chat-v1.0)"
echo " --tp_size N Tensor parallel size (default: 1)"
echo " --transceiver_backend UCX|NIXL Cache-transceiver backend (default: NIXL)"
echo " --transceiver_runtime CPP|PYTHON Cache transceiver runtime (default: CPP)"
echo " --transceiver_runtime CPP|PYTHON Cache transceiver runtime (default: PYTHON)"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

When --transceiver_runtime CPP is selected, should the generated configuration also set kv_cache_config.use_kv_cache_manager_v2: false? The script currently omits this field, and this PR changes its default from auto to True, bypassing the previous V1 fallback for CPP and pairing the C++ transceiver with a V2 manager. Is there another compatibility path that handles this?

echo " --help, -h Show this help message"
exit 0
;;
Expand Down
8 changes: 4 additions & 4 deletions examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -72,8 +72,8 @@ worker_config:
num_slots: 416
layer_updates_per_iter: 1
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
backend: NIXL
transceiver_runtime: PYTHON
max_tokens_in_buffer: 8448
stream_interval: 20
num_postprocess_workers: 4
Expand All @@ -94,7 +94,7 @@ worker_config:
free_gpu_memory_fraction: 0.75
dtype: fp8
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
backend: NIXL
transceiver_runtime: PYTHON
max_tokens_in_buffer: 8448
trust_remote_code: true
3 changes: 3 additions & 0 deletions tensorrt_llm/_torch/pyexecutor/_util.py
Original file line number Diff line number Diff line change
Expand Up @@ -1399,6 +1399,9 @@ def try_prepare_estimation(self) -> bool:
self._skip_est = True
model_config = self._model_engine.model.model_config
if model_config.attn_backend == "VANILLA":
if (self._is_kv_cache_manager_v2
and 'cp_type' not in self._mapping.cp_config):
self._skip_est = True
estimating_kv_cache = False
logger.info(
"KV cache size estimation is not supported for Vanilla attention backend, disable it."
Expand Down
20 changes: 10 additions & 10 deletions tensorrt_llm/llmapi/llm_args.py
Original file line number Diff line number Diff line change
Expand Up @@ -4305,7 +4305,7 @@ class KvCacheConfig(StrictBaseModel, PybindMirror):
description="Configuration for reusable Mamba state snapshots.")

use_kv_cache_manager_v2: bool | Literal["auto"] = Field(
default="auto",
default=True,
status="prototype",
description=
"Whether to use the KV cache manager v2 (experimental). 'auto' uses "
Expand Down Expand Up @@ -4624,17 +4624,17 @@ class CacheTransceiverConfig(StrictBaseModel, PybindMirror):
"The communication backend type to use for the cache transceiver.")

transceiver_runtime: Optional[Literal["CPP", "PYTHON", "auto"]] = Field(
default="auto",
description=
"The runtime implementation. 'auto' (default) adopts the model's "
"preferred runtime when it declares one; otherwise it selects the "
"Python transceiver, falling back to the C++ transceiver only when "
"this config itself rules it out (non-NIXL backend or a null "
"kv_transfer_timeout_ms) — any other incompatibility fails at "
default="PYTHON",
description=
"The runtime implementation. 'PYTHON' (default) selects the Python "
"transceiver, while 'CPP' selects the C++ transceiver. 'auto' adopts "
"the model's preferred runtime when it declares one; otherwise it "
"selects the Python transceiver, falling back to the C++ transceiver "
"only when this config itself rules it out (non-NIXL backend or a "
"null kv_transfer_timeout_ms) — any other incompatibility fails at "
"transceiver creation. The fallback is decided independently on "
"each server and is only logged, not surfaced, so keep context and "
"generation server configurations consistent. 'CPP' selects the C++ "
"transceiver, 'PYTHON' the Python transceiver. None is equivalent "
"generation server configurations consistent. None is equivalent "
"to 'CPP'. 'auto' is resolved on the PyTorch backend's standard "
"model-loading path.")

Expand Down
13 changes: 8 additions & 5 deletions tests/integration/defs/accuracy/test_disaggregated_serving.py
Original file line number Diff line number Diff line change
Expand Up @@ -698,7 +698,7 @@ def run_parallel_test(model_name: str,
test_sets: List[LlmapiAccuracyTestHarness],
ctx_model: str = None,
gen_model: str = None,
cache_transceiver_backend: str = "DEFAULT",
cache_transceiver_backend: str = "NIXL",
trust_remote_code: bool = False,
quant_algo: str = None,
kv_cache_quant_algo: str = None,
Expand Down Expand Up @@ -896,6 +896,7 @@ def test_auto_dtype_with_helix(self, comms_medium, cuda_graph_config,
"enable_block_reuse": False,
"enable_partial_reuse": False,
"tokens_per_block": 32,
"use_kv_cache_manager_v2": False,
}
ctx_server_config = {
"pipeline_parallel_size": 1,
Expand Down Expand Up @@ -1390,7 +1391,7 @@ def test_auto_dtype(self, overlap_scheduler, enable_partial_reuse):
"disable_overlap_scheduler": True,
"cuda_graph_config": None,
"cache_transceiver_config": {
"backend": "DEFAULT",
"backend": "NIXL",
"max_tokens_in_buffer": 4096
},
"kv_cache_config": kv_cache_config,
Expand All @@ -1399,7 +1400,7 @@ def test_auto_dtype(self, overlap_scheduler, enable_partial_reuse):
"disable_overlap_scheduler": overlap_scheduler,
"cuda_graph_config": None,
"cache_transceiver_config": {
"backend": "DEFAULT",
"backend": "NIXL",
"max_tokens_in_buffer": 4096
},
"kv_cache_config": kv_cache_config,
Expand Down Expand Up @@ -1432,7 +1433,7 @@ def _test_chunked_prefill_helper(self, *, ctx_pp: int):
"disable_overlap_scheduler": True,
"cuda_graph_config": None,
"cache_transceiver_config": {
"backend": "DEFAULT",
"backend": "NIXL",
"max_tokens_in_buffer": 4096
},
"enable_chunked_prefill": True,
Expand All @@ -1443,7 +1444,7 @@ def _test_chunked_prefill_helper(self, *, ctx_pp: int):
gen_server_config = {
"cuda_graph_config": None,
"cache_transceiver_config": {
"backend": "DEFAULT",
"backend": "NIXL",
"max_tokens_in_buffer": 4096
},
"max_batch_size": max_batch_size,
Expand Down Expand Up @@ -1487,9 +1488,11 @@ def _run_helix_test(self, comms_medium, cuda_graph_config, gen_pp, gen_tp,
"enable_block_reuse": False,
"enable_partial_reuse": False,
"tokens_per_block": 32,
"use_kv_cache_manager_v2": False,
}
cache_transceiver_config = {
"backend": "DEFAULT",
"transceiver_runtime": "CPP",
"max_tokens_in_buffer": 8192,
}
ctx_server_config = {
Expand Down
Original file line number Diff line number Diff line change
@@ -1,3 +1,6 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

"""DWDP disaggregated serving accuracy tests.

Separated from test_disaggregated_serving.py to isolate MPI-dependent test
Expand Down Expand Up @@ -228,8 +231,7 @@ def test_dwdp_accuracy(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
"backend": "UCX",
"transceiver_runtime": "CPP",
"backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
Expand Down Expand Up @@ -261,8 +263,7 @@ def test_dwdp_accuracy(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
"backend": "UCX",
"transceiver_runtime": "CPP",
"backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
Expand Down Expand Up @@ -345,8 +346,7 @@ def test_dwdp_accuracy_contention_opt(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
"backend": "UCX",
"transceiver_runtime": "CPP",
"backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
Expand Down Expand Up @@ -379,8 +379,7 @@ def test_dwdp_accuracy_contention_opt(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
"backend": "UCX",
"transceiver_runtime": "CPP",
"backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
Expand Down Expand Up @@ -482,8 +481,7 @@ def test_dwdp_accuracy_mode_b_overlap(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
"backend": "UCX",
"transceiver_runtime": "CPP",
"backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
Expand Down Expand Up @@ -519,8 +517,7 @@ def test_dwdp_accuracy_mode_b_overlap(self):
"tokens_per_block": 32,
},
"cache_transceiver_config": {
"backend": "UCX",
"transceiver_runtime": "CPP",
"backend": "NIXL",
"max_tokens_in_buffer": 8192,
},
"moe_config": {
Expand Down

This file was deleted.

Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ context_servers:
enable_block_reuse: false
enable_partial_reuse: false
tokens_per_block: 32
use_kv_cache_manager_v2: false
tensor_parallel_size: 2
pipeline_parallel_size: 1
cache_transceiver_config:
Expand All @@ -28,6 +29,7 @@ generation_servers:
enable_block_reuse: false
enable_partial_reuse: false
tokens_per_block: 32
use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP

This file was deleted.

Original file line number Diff line number Diff line change
Expand Up @@ -16,10 +16,12 @@ context_servers:
kv_cache_config:
enable_block_reuse: false
free_gpu_memory_fraction: 0.3
use_kv_cache_manager_v2: false
disable_overlap_scheduler: true
cuda_graph_config: null
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
# Intentionally small to reproduce buffer overflow bug
max_tokens_in_buffer: 2048
generation_servers:
Expand All @@ -37,9 +39,11 @@ generation_servers:
kv_cache_config:
enable_block_reuse: false
free_gpu_memory_fraction: 0.3
use_kv_cache_manager_v2: false
disable_overlap_scheduler: true
cuda_graph_config: null
cache_transceiver_config:
backend: UCX
transceiver_runtime: CPP
# Intentionally small to reproduce buffer overflow bug
max_tokens_in_buffer: 2048
Original file line number Diff line number Diff line change
@@ -1,3 +1,7 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

# Pin v1 KVCM and the C++ transceiver for v1 timing metrics and CSV output.
hostname: localhost
model: TinyLlama/TinyLlama-1.1B-Chat-v1.0
free_gpu_memory_fraction: 0.25
Expand All @@ -12,6 +16,8 @@ context_servers:
pipeline_parallel_size: 1
return_perf_metrics: True
perf_metrics_max_requests: 1000
kv_cache_config:
use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: DEFAULT
transceiver_runtime: CPP
Expand All @@ -21,6 +27,8 @@ generation_servers:
pipeline_parallel_size: 1
return_perf_metrics: True
perf_metrics_max_requests: 1000
kv_cache_config:
use_kv_cache_manager_v2: false
cache_transceiver_config:
backend: DEFAULT
transceiver_runtime: CPP
Loading
Loading