diff --git a/examples/disaggregated/slurm/service_discovery_example/launch.slurm b/examples/disaggregated/slurm/service_discovery_example/launch.slurm index 60bac45b2e2d..ed584a50533d 100644 --- a/examples/disaggregated/slurm/service_discovery_example/launch.slurm +++ b/examples/disaggregated/slurm/service_discovery_example/launch.slurm @@ -44,16 +44,14 @@ cat >${work_path}/ctx_config.yaml << EOL disable_overlap_scheduler: True internal_request_auth_key: ${internal_request_auth_key} cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP + backend: NIXL max_tokens_in_buffer: 2048 EOL cat >${work_path}/gen_config.yaml << EOL internal_request_auth_key: ${internal_request_auth_key} cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP + backend: NIXL max_tokens_in_buffer: 2048 EOL diff --git a/examples/disaggregated/slurm/simple_example/ctx_extra-llm-api-config.yaml b/examples/disaggregated/slurm/simple_example/ctx_extra-llm-api-config.yaml index cce87b3ea2f1..a43d34f2d485 100644 --- a/examples/disaggregated/slurm/simple_example/ctx_extra-llm-api-config.yaml +++ b/examples/disaggregated/slurm/simple_example/ctx_extra-llm-api-config.yaml @@ -2,6 +2,5 @@ # not yet supported in disaggregated context server architectures. disable_overlap_scheduler: True cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP + backend: NIXL max_tokens_in_buffer: 2048 diff --git a/examples/disaggregated/slurm/simple_example/gen_extra-llm-api-config.yaml b/examples/disaggregated/slurm/simple_example/gen_extra-llm-api-config.yaml index 5324fd439098..61bbe07a7d94 100644 --- a/examples/disaggregated/slurm/simple_example/gen_extra-llm-api-config.yaml +++ b/examples/disaggregated/slurm/simple_example/gen_extra-llm-api-config.yaml @@ -1,4 +1,3 @@ cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP + backend: NIXL max_tokens_in_buffer: 2048 diff --git a/examples/dwdp/reproduce.py b/examples/dwdp/reproduce.py index f214329ebf7d..959e4f535805 100644 --- a/examples/dwdp/reproduce.py +++ b/examples/dwdp/reproduce.py @@ -220,8 +220,7 @@ def build_worker_config(experiment: Dict[str, Any]) -> Dict[str, Any]: "backend": "CUTEDSL", }, "cache_transceiver_config": { - "backend": "UCX", - "transceiver_runtime": "CPP", + "backend": "NIXL", "max_tokens_in_buffer": max_tokens_in_buffer, }, "num_postprocess_workers": 4, @@ -256,8 +255,7 @@ def build_worker_config(experiment: Dict[str, Any]) -> Dict[str, Any]: "free_gpu_memory_fraction": 0.3, }, "cache_transceiver_config": { - "backend": "UCX", - "transceiver_runtime": "CPP", + "backend": "NIXL", "max_tokens_in_buffer": max_tokens_in_buffer, }, "moe_config": { diff --git a/examples/llm-api/quickstart_advanced.py b/examples/llm-api/quickstart_advanced.py index a167ec0ae3e7..371deac1afa1 100644 --- a/examples/llm-api/quickstart_advanced.py +++ b/examples/llm-api/quickstart_advanced.py @@ -151,11 +151,11 @@ def add_llm_args(parser): action='store_true') parser.add_argument( '--use_kv_cache_manager_v2', - default='auto', + default=True, type=_parse_kv_cache_manager_v2, metavar='{auto,true,false}', help= - 'Whether to use KVCacheManagerV2 for KV cache management (PyTorch backend). Defaults to model-specific auto selection.', + 'Whether to use KVCacheManagerV2 for KV cache management (PyTorch backend). Defaults to true; use auto for model-specific selection.', ) # Runtime diff --git a/examples/ray_orchestrator/disaggregated/disagg_serving_local.sh b/examples/ray_orchestrator/disaggregated/disagg_serving_local.sh index 40ee6745e693..a49e2f404b4a 100644 --- a/examples/ray_orchestrator/disaggregated/disagg_serving_local.sh +++ b/examples/ray_orchestrator/disaggregated/disagg_serving_local.sh @@ -8,7 +8,7 @@ ATTACH_MODE=false MODEL_DIR="TinyLlama/TinyLlama-1.1B-Chat-v1.0" TP_SIZE=1 TRANSCEIVER_BACKEND="NIXL" -TRANSCEIVER_RUNTIME="CPP" +TRANSCEIVER_RUNTIME="PYTHON" USAGE="Usage: $0 [--executor ray|mpi] [--attach] [--model model_dir] [--tp_size N] [--transceiver_backend UCX|NIXL] [--transceiver_runtime CPP|PYTHON] [--help]" while [[ $# -gt 0 ]]; do @@ -45,7 +45,7 @@ while [[ $# -gt 0 ]]; do echo " --model model_dir Model directory (default: TinyLlama/TinyLlama-1.1B-Chat-v1.0)" echo " --tp_size N Tensor parallel size (default: 1)" echo " --transceiver_backend UCX|NIXL Cache-transceiver backend (default: NIXL)" - echo " --transceiver_runtime CPP|PYTHON Cache transceiver runtime (default: CPP)" + echo " --transceiver_runtime CPP|PYTHON Cache transceiver runtime (default: PYTHON)" echo " --help, -h Show this help message" exit 0 ;; diff --git a/examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml b/examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml index 17e0f96a381f..789e3ad847ed 100644 --- a/examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml +++ b/examples/wide_ep/slurm_scripts/kimi-k2-thinking.yaml @@ -72,8 +72,8 @@ worker_config: num_slots: 416 layer_updates_per_iter: 1 cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP + backend: NIXL + transceiver_runtime: PYTHON max_tokens_in_buffer: 8448 stream_interval: 20 num_postprocess_workers: 4 @@ -94,7 +94,7 @@ worker_config: free_gpu_memory_fraction: 0.75 dtype: fp8 cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP + backend: NIXL + transceiver_runtime: PYTHON max_tokens_in_buffer: 8448 trust_remote_code: true diff --git a/tensorrt_llm/_torch/pyexecutor/_util.py b/tensorrt_llm/_torch/pyexecutor/_util.py index 660ffe11b601..73759bfb55ab 100644 --- a/tensorrt_llm/_torch/pyexecutor/_util.py +++ b/tensorrt_llm/_torch/pyexecutor/_util.py @@ -1399,6 +1399,9 @@ def try_prepare_estimation(self) -> bool: self._skip_est = True model_config = self._model_engine.model.model_config if model_config.attn_backend == "VANILLA": + if (self._is_kv_cache_manager_v2 + and 'cp_type' not in self._mapping.cp_config): + self._skip_est = True estimating_kv_cache = False logger.info( "KV cache size estimation is not supported for Vanilla attention backend, disable it." diff --git a/tensorrt_llm/llmapi/llm_args.py b/tensorrt_llm/llmapi/llm_args.py index d596c1d38a4e..e3aa05cd3a9f 100644 --- a/tensorrt_llm/llmapi/llm_args.py +++ b/tensorrt_llm/llmapi/llm_args.py @@ -4305,7 +4305,7 @@ class KvCacheConfig(StrictBaseModel, PybindMirror): description="Configuration for reusable Mamba state snapshots.") use_kv_cache_manager_v2: bool | Literal["auto"] = Field( - default="auto", + default=True, status="prototype", description= "Whether to use the KV cache manager v2 (experimental). 'auto' uses " @@ -4624,17 +4624,17 @@ class CacheTransceiverConfig(StrictBaseModel, PybindMirror): "The communication backend type to use for the cache transceiver.") transceiver_runtime: Optional[Literal["CPP", "PYTHON", "auto"]] = Field( - default="auto", - description= - "The runtime implementation. 'auto' (default) adopts the model's " - "preferred runtime when it declares one; otherwise it selects the " - "Python transceiver, falling back to the C++ transceiver only when " - "this config itself rules it out (non-NIXL backend or a null " - "kv_transfer_timeout_ms) — any other incompatibility fails at " + default="PYTHON", + description= + "The runtime implementation. 'PYTHON' (default) selects the Python " + "transceiver, while 'CPP' selects the C++ transceiver. 'auto' adopts " + "the model's preferred runtime when it declares one; otherwise it " + "selects the Python transceiver, falling back to the C++ transceiver " + "only when this config itself rules it out (non-NIXL backend or a " + "null kv_transfer_timeout_ms) — any other incompatibility fails at " "transceiver creation. The fallback is decided independently on " "each server and is only logged, not surfaced, so keep context and " - "generation server configurations consistent. 'CPP' selects the C++ " - "transceiver, 'PYTHON' the Python transceiver. None is equivalent " + "generation server configurations consistent. None is equivalent " "to 'CPP'. 'auto' is resolved on the PyTorch backend's standard " "model-loading path.") diff --git a/tests/integration/defs/accuracy/test_disaggregated_serving.py b/tests/integration/defs/accuracy/test_disaggregated_serving.py index d6f8287c998e..7e3512721ed3 100644 --- a/tests/integration/defs/accuracy/test_disaggregated_serving.py +++ b/tests/integration/defs/accuracy/test_disaggregated_serving.py @@ -698,7 +698,7 @@ def run_parallel_test(model_name: str, test_sets: List[LlmapiAccuracyTestHarness], ctx_model: str = None, gen_model: str = None, - cache_transceiver_backend: str = "DEFAULT", + cache_transceiver_backend: str = "NIXL", trust_remote_code: bool = False, quant_algo: str = None, kv_cache_quant_algo: str = None, @@ -896,6 +896,7 @@ def test_auto_dtype_with_helix(self, comms_medium, cuda_graph_config, "enable_block_reuse": False, "enable_partial_reuse": False, "tokens_per_block": 32, + "use_kv_cache_manager_v2": False, } ctx_server_config = { "pipeline_parallel_size": 1, @@ -1390,7 +1391,7 @@ def test_auto_dtype(self, overlap_scheduler, enable_partial_reuse): "disable_overlap_scheduler": True, "cuda_graph_config": None, "cache_transceiver_config": { - "backend": "DEFAULT", + "backend": "NIXL", "max_tokens_in_buffer": 4096 }, "kv_cache_config": kv_cache_config, @@ -1399,7 +1400,7 @@ def test_auto_dtype(self, overlap_scheduler, enable_partial_reuse): "disable_overlap_scheduler": overlap_scheduler, "cuda_graph_config": None, "cache_transceiver_config": { - "backend": "DEFAULT", + "backend": "NIXL", "max_tokens_in_buffer": 4096 }, "kv_cache_config": kv_cache_config, @@ -1432,7 +1433,7 @@ def _test_chunked_prefill_helper(self, *, ctx_pp: int): "disable_overlap_scheduler": True, "cuda_graph_config": None, "cache_transceiver_config": { - "backend": "DEFAULT", + "backend": "NIXL", "max_tokens_in_buffer": 4096 }, "enable_chunked_prefill": True, @@ -1443,7 +1444,7 @@ def _test_chunked_prefill_helper(self, *, ctx_pp: int): gen_server_config = { "cuda_graph_config": None, "cache_transceiver_config": { - "backend": "DEFAULT", + "backend": "NIXL", "max_tokens_in_buffer": 4096 }, "max_batch_size": max_batch_size, @@ -1487,9 +1488,11 @@ def _run_helix_test(self, comms_medium, cuda_graph_config, gen_pp, gen_tp, "enable_block_reuse": False, "enable_partial_reuse": False, "tokens_per_block": 32, + "use_kv_cache_manager_v2": False, } cache_transceiver_config = { "backend": "DEFAULT", + "transceiver_runtime": "CPP", "max_tokens_in_buffer": 8192, } ctx_server_config = { diff --git a/tests/integration/defs/accuracy/test_dwdp_disaggregated_serving.py b/tests/integration/defs/accuracy/test_dwdp_disaggregated_serving.py index 801a6858706d..eab54355d695 100644 --- a/tests/integration/defs/accuracy/test_dwdp_disaggregated_serving.py +++ b/tests/integration/defs/accuracy/test_dwdp_disaggregated_serving.py @@ -1,3 +1,6 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + """DWDP disaggregated serving accuracy tests. Separated from test_disaggregated_serving.py to isolate MPI-dependent test @@ -228,8 +231,7 @@ def test_dwdp_accuracy(self): "tokens_per_block": 32, }, "cache_transceiver_config": { - "backend": "UCX", - "transceiver_runtime": "CPP", + "backend": "NIXL", "max_tokens_in_buffer": 8192, }, "moe_config": { @@ -261,8 +263,7 @@ def test_dwdp_accuracy(self): "tokens_per_block": 32, }, "cache_transceiver_config": { - "backend": "UCX", - "transceiver_runtime": "CPP", + "backend": "NIXL", "max_tokens_in_buffer": 8192, }, "moe_config": { @@ -345,8 +346,7 @@ def test_dwdp_accuracy_contention_opt(self): "tokens_per_block": 32, }, "cache_transceiver_config": { - "backend": "UCX", - "transceiver_runtime": "CPP", + "backend": "NIXL", "max_tokens_in_buffer": 8192, }, "moe_config": { @@ -379,8 +379,7 @@ def test_dwdp_accuracy_contention_opt(self): "tokens_per_block": 32, }, "cache_transceiver_config": { - "backend": "UCX", - "transceiver_runtime": "CPP", + "backend": "NIXL", "max_tokens_in_buffer": 8192, }, "moe_config": { @@ -482,8 +481,7 @@ def test_dwdp_accuracy_mode_b_overlap(self): "tokens_per_block": 32, }, "cache_transceiver_config": { - "backend": "UCX", - "transceiver_runtime": "CPP", + "backend": "NIXL", "max_tokens_in_buffer": 8192, }, "moe_config": { @@ -519,8 +517,7 @@ def test_dwdp_accuracy_mode_b_overlap(self): "tokens_per_block": 32, }, "cache_transceiver_config": { - "backend": "UCX", - "transceiver_runtime": "CPP", + "backend": "NIXL", "max_tokens_in_buffer": 8192, }, "moe_config": { diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml deleted file mode 100644 index e43517965b8c..000000000000 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp1_gentp1_deepseek_v3_lite_ucx.yaml +++ /dev/null @@ -1,20 +0,0 @@ -hostname: localhost -model: DeepSeek-V3-Lite/fp8 -free_gpu_memory_fraction: 0.1 -backend: pytorch -cuda_graph_config: null -disable_overlap_scheduler: true -context_servers: - num_instances: 1 - tensor_parallel_size: 1 - pipeline_parallel_size: 1 - cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP -generation_servers: - num_instances: 1 - tensor_parallel_size: 1 - pipeline_parallel_size: 1 - cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml index 9fa5a4a5a1b0..87a04bde64f1 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp1cp2_deepseek_v3_lite_bf16_tllm_gen.yaml @@ -10,6 +10,7 @@ context_servers: enable_block_reuse: false enable_partial_reuse: false tokens_per_block: 32 + use_kv_cache_manager_v2: false tensor_parallel_size: 2 pipeline_parallel_size: 1 cache_transceiver_config: @@ -28,6 +29,7 @@ generation_servers: enable_block_reuse: false enable_partial_reuse: false tokens_per_block: 32 + use_kv_cache_manager_v2: false cache_transceiver_config: backend: UCX transceiver_runtime: CPP diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml deleted file mode 100644 index 689f7d53f64b..000000000000 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml +++ /dev/null @@ -1,19 +0,0 @@ -hostname: localhost -model: DeepSeek-V3-Lite/fp8 -free_gpu_memory_fraction: 0.25 -backend: pytorch -disable_overlap_scheduler: true -context_servers: - num_instances: 1 - tensor_parallel_size: 2 - pipeline_parallel_size: 1 - cache_transceiver_config: - backend: MPI - transceiver_runtime: CPP -generation_servers: - num_instances: 1 - tensor_parallel_size: 2 - pipeline_parallel_size: 1 - cache_transceiver_config: - backend: MPI - transceiver_runtime: CPP diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml index fa65a710981c..ee04da6504ed 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_llama4_kv_cache_overflow.yaml @@ -16,10 +16,12 @@ context_servers: kv_cache_config: enable_block_reuse: false free_gpu_memory_fraction: 0.3 + use_kv_cache_manager_v2: false disable_overlap_scheduler: true cuda_graph_config: null cache_transceiver_config: backend: UCX + transceiver_runtime: CPP # Intentionally small to reproduce buffer overflow bug max_tokens_in_buffer: 2048 generation_servers: @@ -37,9 +39,11 @@ generation_servers: kv_cache_config: enable_block_reuse: false free_gpu_memory_fraction: 0.3 + use_kv_cache_manager_v2: false disable_overlap_scheduler: true cuda_graph_config: null cache_transceiver_config: backend: UCX + transceiver_runtime: CPP # Intentionally small to reproduce buffer overflow bug max_tokens_in_buffer: 2048 diff --git a/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml b/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml index a15f68e6381c..7d5a9abadf00 100644 --- a/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml +++ b/tests/integration/defs/disaggregated/test_configs/disagg_config_metrics.yaml @@ -1,3 +1,7 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Pin v1 KVCM and the C++ transceiver for v1 timing metrics and CSV output. hostname: localhost model: TinyLlama/TinyLlama-1.1B-Chat-v1.0 free_gpu_memory_fraction: 0.25 @@ -12,6 +16,8 @@ context_servers: pipeline_parallel_size: 1 return_perf_metrics: True perf_metrics_max_requests: 1000 + kv_cache_config: + use_kv_cache_manager_v2: false cache_transceiver_config: backend: DEFAULT transceiver_runtime: CPP @@ -21,6 +27,8 @@ generation_servers: pipeline_parallel_size: 1 return_perf_metrics: True perf_metrics_max_requests: 1000 + kv_cache_config: + use_kv_cache_manager_v2: false cache_transceiver_config: backend: DEFAULT transceiver_runtime: CPP diff --git a/tests/integration/defs/disaggregated/test_disaggregated.py b/tests/integration/defs/disaggregated/test_disaggregated.py index 163f05c7358b..24e3f4d4cf45 100644 --- a/tests/integration/defs/disaggregated/test_disaggregated.py +++ b/tests/integration/defs/disaggregated/test_disaggregated.py @@ -346,8 +346,6 @@ def get_test_config(test_desc, example_dir, test_root): f"{test_configs_root}/disagg_config_ctxpp4_genpp4.yaml", "ctxpp4_gentp4": f"{test_configs_root}/disagg_config_ctxpp4_gentp4.yaml", - "deepseek_v3_lite_fp8_mpi": - f"{test_configs_root}/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_mpi.yaml", "deepseek_v3_lite_fp8_nixl": f"{test_configs_root}/disagg_config_ctxtp2_gentp2_deepseek_v3_lite_nixl.yaml", "deepseek_v3_lite_fp8_tp1": @@ -1800,28 +1798,6 @@ def test_disaggregated_ctxpp4_gentp4(disaggregated_test_root, llm_venv, cwd=llm_venv.get_working_directory()) -@skip_no_hopper -@pytest.mark.skip_less_device(4) -@pytest.mark.skip( - reason="MPI cache transceiver requires shared MPI process group, " - "incompatible with service discovery which launches separate subprocesses") -@pytest.mark.parametrize("deepseek_v3_model_root", ['DeepSeek-V3-Lite-fp8'], - indirect=True) -def test_disaggregated_deepseek_v3_lite_fp8_mpi(disaggregated_test_root, - disaggregated_example_root, - llm_venv, - deepseek_v3_model_root): - setup_model_symlink(llm_venv, deepseek_v3_model_root, - "DeepSeek-V3-Lite/fp8") - env = llm_venv._new_env.copy() - env["TRTLLM_USE_MPI_KVCACHE"] = "1" - run_disaggregated_test(disaggregated_example_root, - "deepseek_v3_lite_fp8_mpi", - env=env, - model_path=deepseek_v3_model_root, - cwd=llm_venv.get_working_directory()) - - @skip_no_hopper @pytest.mark.parametrize("deepseek_v3_model_root", ['DeepSeek-V3-Lite-fp8'], indirect=True) diff --git a/tests/integration/defs/disaggregated/test_disaggregated_etcd.py b/tests/integration/defs/disaggregated/test_disaggregated_etcd.py index ec7956ed7393..2e46a2c03a73 100644 --- a/tests/integration/defs/disaggregated/test_disaggregated_etcd.py +++ b/tests/integration/defs/disaggregated/test_disaggregated_etcd.py @@ -88,7 +88,7 @@ def start_context_server(config, server_env = env.copy() if env else os.environ.copy() server_env["CUDA_VISIBLE_DEVICES"] = str(gpu_id) - server_env["TRTLLM_USE_UCX_KVCACHE"] = "1" + server_env["TRTLLM_USE_NIXL_KVCACHE"] = "1" server_env["UCX_TLS"] = get_ucx_tls() logger.info(f"Starting CONTEXT server on GPU {gpu_id} (port {port})...") @@ -115,7 +115,7 @@ def start_generation_server(config, server_env = env.copy() if env else os.environ.copy() server_env["CUDA_VISIBLE_DEVICES"] = str(gpu_id) - server_env["TRTLLM_USE_UCX_KVCACHE"] = "1" + server_env["TRTLLM_USE_NIXL_KVCACHE"] = "1" server_env["UCX_TLS"] = get_ucx_tls() logger.info(f"Starting GENERATION server on GPU {gpu_id} (port {port})...") diff --git a/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py b/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py index da8aaeefd6ea..99279bbdd534 100644 --- a/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py +++ b/tests/integration/defs/disaggregated/test_disaggregated_single_gpu.py @@ -1,3 +1,6 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + import asyncio import os import pickle @@ -1050,8 +1053,9 @@ def test_arbitrary_kv_cache_transfer(model, generation_overlap): cuda_graph_config=CudaGraphConfig())) kv_cache_configs = [ - KvCacheConfig(max_tokens=2048 * 8, enable_block_reuse=True) - for _ in range(2) + KvCacheConfig(max_tokens=2048 * 8, + enable_block_reuse=True, + use_kv_cache_manager_v2=False) for _ in range(2) ] # Arbitrary transfer uses the C++ serialized DataTransceiverState protocol. cache_transceiver_configs = [ @@ -1210,8 +1214,9 @@ def test_arbitrary_kv_cache_transfer_missing_blocks(model, generation_overlap): cuda_graph_config=CudaGraphConfig())) kv_cache_configs = [ - KvCacheConfig(max_tokens=2048 * 8, enable_block_reuse=True) - for _ in range(2) + KvCacheConfig(max_tokens=2048 * 8, + enable_block_reuse=True, + use_kv_cache_manager_v2=False) for _ in range(2) ] # Arbitrary transfer uses the C++ serialized DataTransceiverState protocol. cache_transceiver_configs = [ diff --git a/tests/integration/defs/examples/test_ray.py b/tests/integration/defs/examples/test_ray.py index 201192d5fdd1..5023fa8d99de 100644 --- a/tests/integration/defs/examples/test_ray.py +++ b/tests/integration/defs/examples/test_ray.py @@ -70,12 +70,6 @@ def test_llm_inference_distributed_ray(ray_example_root, llm_venv, tp_size, venv_check_call(llm_venv, cmd) -@pytest.mark.skip_less_device(2) -@pytest.mark.parametrize("tp_size", [1, 2], ids=["tp1", "tp2"]) -def test_ray_disaggregated_serving(ray_example_root, llm_venv, tp_size): - _run_ray_disaggregated_serving(ray_example_root, tp_size, "NIXL", "CPP") - - @pytest.mark.skip_less_device(2) @pytest.mark.parametrize("tp_size", [1, 2], ids=["tp1", "tp2"]) def test_ray_disaggregated_serving_python(ray_example_root, llm_venv, tp_size): diff --git a/tests/integration/test_lists/qa/llm_function_core.txt b/tests/integration/test_lists/qa/llm_function_core.txt index 798a9e4da53c..e014ba906f89 100644 --- a/tests/integration/test_lists/qa/llm_function_core.txt +++ b/tests/integration/test_lists/qa/llm_function_core.txt @@ -695,7 +695,6 @@ disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_att disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxpp2_gentp2_one_mtp[DeepSeek-V3-Lite-fp8] disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxtp2ep2pp2_gentp4_one_mtp_block_reuse[DeepSeek-V3-Lite-fp8] disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_ctxtp2ep2pp2_gentp4_one_mtp_block_reuse_long_prompt[DeepSeek-V3-Lite-fp8] -disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_mpi[DeepSeek-V3-Lite-fp8] disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_nixl[DeepSeek-V3-Lite-fp8] disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_overlap_cuda_graph[DeepSeek-V3-Lite-fp8] disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_tp1_attention_dp_overlap_one_mtp[DeepSeek-V3-Lite-fp8] diff --git a/tests/integration/test_lists/test-db/l0_dgx_b200.yml b/tests/integration/test_lists/test-db/l0_dgx_b200.yml index 91ac09081e6b..a194d0175b86 100644 --- a/tests/integration/test_lists/test-db/l0_dgx_b200.yml +++ b/tests/integration/test_lists/test-db/l0_dgx_b200.yml @@ -144,7 +144,6 @@ l0_dgx_b200: - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp2_gentp2[TinyLlama-1.1B-Chat-v1.0] - disaggregated/test_disaggregated.py::test_disaggregated_ctxpp4_gentp4[TinyLlama-1.1B-Chat-v1.0] - examples/test_ray.py::test_llm_inference_distributed_ray[tp2pp2] - - examples/test_ray.py::test_ray_disaggregated_serving[tp2] - examples/test_ray.py::test_ray_disaggregated_serving_python[tp2] - condition: ranges: diff --git a/tests/integration/test_lists/test-db/l0_dgx_h100.yml b/tests/integration/test_lists/test-db/l0_dgx_h100.yml index 7279dc0b1a7d..563a78b8d8f5 100644 --- a/tests/integration/test_lists/test-db/l0_dgx_h100.yml +++ b/tests/integration/test_lists/test-db/l0_dgx_h100.yml @@ -198,7 +198,6 @@ l0_dgx_h100: - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales_cuda_graph_padding_4gpus[attention_dp=True-mtp_nextn=0] - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales_cuda_graph_padding_4gpus[attention_dp=True-mtp_nextn=2] - accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_guided_decoding_4gpus[xgrammar-mtp_nextn=2] - - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_mpi[DeepSeek-V3-Lite-fp8] - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_nixl[DeepSeek-V3-Lite-fp8] - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_attention_dp[DeepSeek-V3-Lite-fp8] - disaggregated/test_disaggregated.py::test_disaggregated_deepseek_v3_lite_fp8_attention_dp_overlap[DeepSeek-V3-Lite-fp8] @@ -292,7 +291,6 @@ l0_dgx_h100: - examples/test_ray.py::test_llm_inference_distributed_ray[tp2] - examples/test_ray.py::test_llm_inference_distributed_ray[pp2] - examples/test_ray.py::test_llm_inference_distributed_ray[tep2] - - examples/test_ray.py::test_ray_disaggregated_serving[tp1] - examples/test_ray.py::test_ray_disaggregated_serving_python[tp1] - condition: ranges: diff --git a/tests/integration/test_lists/waives.txt b/tests/integration/test_lists/waives.txt index 49833d85fb38..dce7568ed227 100644 --- a/tests/integration/test_lists/waives.txt +++ b/tests/integration/test_lists/waives.txt @@ -85,6 +85,11 @@ examples/visual_gen/test_visual_gen_wan.py::test_wan_feature_accuracy_against_go examples/visual_gen/test_visual_gen_wan.py::test_wan_feature_accuracy_against_golden[wan22-cuda-graph] SKIP (https://nvbugs/6572800) examples/visual_gen/test_visual_gen_wan.py::test_wan_feature_accuracy_against_golden[wan22-fp8-blockwise] SKIP (https://nvbugs/6572800) examples/visual_gen/test_visual_gen_wan.py::test_wan_feature_accuracy_against_golden[wan22-nvfp4] SKIP (https://nvbugs/6572800) +full:A10/test_e2e.py::test_openai_chat_multimodal_example SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213) +full:A10/test_e2e.py::test_trtllm_serve_multimodal_example SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213) +full:A10/unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_image SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213) +full:A10/unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_image_streaming SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213) +full:A10/unittest/grpc/smg/test_smg.py::TestGrpcMultimodalEndToEnd::test_generate_with_rgba_image SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213) full:A100/accuracy/test_llm_api_pytorch.py::TestNemotronV3Super::test_nvfp4_4gpus_hopper_w4a16 SKIP (https://nvbugs/6802472) full:A100/accuracy/test_llm_api_pytorch_multimodal.py::TestExaone4_5_33B::test_auto_dtype[forced_chunked_prefill] SKIP (https://nvbugs/6597570) full:A100/accuracy/test_llm_api_pytorch_multimodal.py::TestExaone4_5_33B::test_auto_dtype[full_budget] SKIP (https://nvbugs/6597570) @@ -119,9 +124,11 @@ full:B300/disaggregated/test_disaggregated.py::test_disaggregated_overlap_gen_fi full:B300/disaggregated/test_disaggregated.py::test_disaggregated_qwen3_32b_fp8[Qwen3/Qwen3-32B-FP8] SKIP (https://nvbugs/6770977) full:B300/kv_cache/test_prefix_aware_scheduling.py::TestServePrefixAwareScheduling::test_multi_round_qa_shared_prefix_smoke SKIP (https://nvbugs/6782589) full:DGX_B200/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_bfloat16[mtp_nextn=2-attention_dp=True-cuda_graph=True-overlap_scheduler=True-torch_compile=False-enable_chunked_prefill=False-v2_kv_cache=True] SKIP (https://nvbugs/6633268) +full:DGX_B200/accuracy/test_llm_api_pytorch.py::TestSeedOss_36B::test_auto_dtype SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213) full:DGX_B200/unittest/tools/test_layer_wise_benchmarks.py::test_performance_alignment[1] SKIP (https://nvbugs/6669275) full:DGX_H100/accuracy/test_disaggregated_serving.py::TestQwen3_5_4B::test_mismatched_block_reuse SKIP (https://nvbugs/6793949) full:DGX_H100/accuracy/test_llm_api_pytorch.py::TestDeepSeekV3Lite::test_fp8_block_scales[mtp=disable-fp8kv=True-attention_dp=False-cuda_graph=True-overlap_scheduler=True-torch_compile=True] SKIP (https://nvbugs/6700265) +full:DGX_H100/accuracy/test_llm_api_pytorch_multimodal.py::TestMistralSmall24B::test_auto_dtype[forced_chunked_prefill] SKIP (KVCM V2 initialization OOM; pending https://github.com/NVIDIA/TensorRT-LLM/pull/19213) full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy SKIP (https://nvbugs/6276923) full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_contention_opt SKIP (https://nvbugs/6276923) full:GB200/accuracy/test_dwdp_disaggregated_serving.py::TestDwdpDeepSeekV3Lite::test_dwdp_accuracy_mode_b_overlap SKIP (https://nvbugs/6276923) diff --git a/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml index f8e513756d32..6a0a0c9c2bc1 100644 --- a/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml +++ b/tests/scripts/perf-sanity/aggregated/dynamo_deepseek_v32_fp4_2_nodes_grace_blackwell.yaml @@ -37,10 +37,6 @@ server_configs: enable_block_reuse: false free_gpu_memory_fraction: 0.9 tokens_per_block: 64 - cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP - max_tokens_in_buffer: 120000 client_configs: - name: "con4_iter10_8k1k" concurrency: 4 diff --git a/tests/scripts/perf-sanity/aggregated/dynamo_k25_thinking_fp4_blackwell.yaml b/tests/scripts/perf-sanity/aggregated/dynamo_k25_thinking_fp4_blackwell.yaml index 2233f5040482..50728425ebad 100644 --- a/tests/scripts/perf-sanity/aggregated/dynamo_k25_thinking_fp4_blackwell.yaml +++ b/tests/scripts/perf-sanity/aggregated/dynamo_k25_thinking_fp4_blackwell.yaml @@ -26,8 +26,8 @@ server_configs: dtype: 'fp8' free_gpu_memory_fraction: 0.75 cache_transceiver_config: - backend: UCX - transceiver_runtime: CPP + backend: NIXL + transceiver_runtime: PYTHON max_tokens_in_buffer: 8448 client_configs: - name: "con128_iter5_2k1k" diff --git a/tests/scripts/perf-sanity/cache_transceiver_precheck/precheck_config.py b/tests/scripts/perf-sanity/cache_transceiver_precheck/precheck_config.py index 497edae4377d..970a254abf9b 100644 --- a/tests/scripts/perf-sanity/cache_transceiver_precheck/precheck_config.py +++ b/tests/scripts/perf-sanity/cache_transceiver_precheck/precheck_config.py @@ -449,7 +449,7 @@ def resolve_plan(cfg, benchmark_mode="e2e"): # True/False from the yaml wins; absent means "auto", which the # driver resolves against the model class's manager preference at # runtime, exactly like serving (_resolve_kv_cache_manager_v2_auto). - plan[f"{role}_use_kv_cache_manager_v2"] = kv_cfg.get("use_kv_cache_manager_v2", "auto") + plan[f"{role}_use_kv_cache_manager_v2"] = kv_cfg.get("use_kv_cache_manager_v2", True) plan["fingerprint"] = plan_fingerprint(plan) return plan diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml index d5046594c038..384f198ca17e 100644 --- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con2048_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml @@ -57,7 +57,6 @@ worker_config: enable_padding: true max_batch_size: 1536 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -81,7 +80,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml index ebc2bac0df08..f18b38e6dc45 100644 --- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml @@ -57,7 +57,6 @@ worker_config: enable_padding: true max_batch_size: 1536 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -81,7 +80,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml index 8cafded2577a..685171bf899e 100644 --- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_1k1k_con64_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml @@ -53,7 +53,6 @@ worker_config: enable_padding: true max_batch_size: 256 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -78,7 +77,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml index 1f6ad52274b9..97ca603485ce 100644 --- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml @@ -65,7 +65,6 @@ worker_config: enable_padding: true max_batch_size: 1280 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.85 dtype: fp8 @@ -94,7 +93,6 @@ worker_config: enable_padding: true max_batch_size: 30 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.8 dtype: fp8 diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml index 45f8e986fd98..c031ad48ecbd 100644 --- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml @@ -53,7 +53,6 @@ worker_config: enable_padding: true max_batch_size: 1024 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -78,7 +77,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml index c2eb718be285..11b64deb265e 100644 --- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml @@ -53,7 +53,6 @@ worker_config: enable_padding: true max_batch_size: 1024 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -78,7 +77,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml index 27aecee14739..b6bde7daf105 100644 --- a/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf-sanity/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml @@ -57,7 +57,6 @@ worker_config: enable_padding: true max_batch_size: 512 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -81,7 +80,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf-sanity/disaggregated/gb200_llama-3.1-8b-bf16_1k1k_con256_ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf-sanity/disaggregated/gb200_llama-3.1-8b-bf16_1k1k_con256_ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml index cc3db00acce4..38cbadd729c7 100644 --- a/tests/scripts/perf-sanity/disaggregated/gb200_llama-3.1-8b-bf16_1k1k_con256_ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf-sanity/disaggregated/gb200_llama-3.1-8b-bf16_1k1k_con256_ctx1_tp1_gen1_tp1_eplb0_mtp0_ccb-NIXL.yaml @@ -68,7 +68,6 @@ worker_config: enable_padding: true max_batch_size: 256 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.85 dtype: auto @@ -90,7 +89,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.85 dtype: auto diff --git a/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml b/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml index bc277d40c127..a055fcfbb2d9 100644 --- a/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml +++ b/tests/scripts/perf-sanity/disaggregated/h200_nemotron-super-fp8_8k1k_con64_ctx1_tp2_gen1_tp2_eplb0_mtp0_ccb-UCX.yaml @@ -59,6 +59,7 @@ worker_config: max_batch_size: 16 kv_cache_config: enable_block_reuse: false + use_kv_cache_manager_v2: false free_gpu_memory_fraction: 0.85 moe_config: backend: TRTLLM @@ -82,6 +83,7 @@ worker_config: max_batch_size: 16 kv_cache_config: enable_block_reuse: false + use_kv_cache_manager_v2: false free_gpu_memory_fraction: 0.85 moe_config: backend: TRTLLM diff --git a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml index ffa8cdfa4522..94f9074fb86d 100644 --- a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con1024_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml @@ -65,7 +65,6 @@ worker_config: enable_padding: true max_batch_size: 1280 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.85 dtype: fp8 @@ -94,7 +93,6 @@ worker_config: enable_padding: true max_batch_size: 30 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.8 dtype: fp8 diff --git a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml index 1d16d05a1ad3..b402a1ddaeb4 100644 --- a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con128_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml @@ -64,7 +64,6 @@ worker_config: enable_padding: true max_batch_size: 1024 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -89,7 +88,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml index 9604ca2743f9..d29d122474e7 100644 --- a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con4_ctx1_tp1_gen1_tp4_eplb0_mtp0_ccb-NIXL.yaml @@ -64,7 +64,6 @@ worker_config: enable_padding: true max_batch_size: 1024 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -89,7 +88,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml index 6740b7f581ec..005fef248f59 100644 --- a/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/gb200_gpt-oss-120b-fp4_8k1k_con512_ctx1_tp1_gen1_dep2_eplb0_mtp0_ccb-NIXL.yaml @@ -68,7 +68,6 @@ worker_config: enable_padding: true max_batch_size: 512 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -92,7 +91,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/scripts/perf/disaggregated/gb200_stress-gpt-oss-120b-fp4_8k1k_ctx1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml b/tests/scripts/perf/disaggregated/gb200_stress-gpt-oss-120b-fp4_8k1k_ctx1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml index f7c14976709b..7b97ad0e0589 100644 --- a/tests/scripts/perf/disaggregated/gb200_stress-gpt-oss-120b-fp4_8k1k_ctx1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml +++ b/tests/scripts/perf/disaggregated/gb200_stress-gpt-oss-120b-fp4_8k1k_ctx1_tp1_gen1_tp4_eplb0_eagle3_ccb-NIXL.yaml @@ -70,7 +70,6 @@ worker_config: enable_padding: true max_batch_size: 1024 kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 @@ -99,7 +98,6 @@ worker_config: enable_attention_dp: false cuda_graph_config: null kv_cache_config: - use_kv_cache_manager_v2: false enable_block_reuse: false free_gpu_memory_fraction: 0.9 dtype: fp8 diff --git a/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py b/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py index 03aeafb8b328..d1386439b730 100644 --- a/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py +++ b/tests/unittest/_torch/disaggregation/test_disagg_inflight_cancel_gate.py @@ -488,7 +488,7 @@ def test_direct_cpp_wrapper_rejects_python_runtime_opt_in(monkeypatch): def test_flag_unset_preserves_existing_backend_selection(monkeypatch): - config = CacheTransceiverConfig(backend="UCX") + config = CacheTransceiverConfig(backend="UCX", transceiver_runtime="CPP") expected = object() constructor = Mock(return_value=expected) monkeypatch.setattr(transceiver_module, "BindKvCacheTransceiver", constructor) @@ -568,7 +568,7 @@ def test_cpp_runtime_keeps_cpp_mamba_manager(monkeypatch, runtime): def test_flag_unset_preserves_libfabric_selection(monkeypatch): monkeypatch.setenv(transceiver_module._NIXL_KVCACHE_BACKEND_ENV, "LIBFABRIC") - config = CacheTransceiverConfig(backend="NIXL") + config = CacheTransceiverConfig(backend="NIXL", transceiver_runtime="CPP") expected = object() constructor = Mock(return_value=expected) monkeypatch.setattr(transceiver_module, "BindKvCacheTransceiver", constructor) @@ -590,7 +590,7 @@ def test_flag_unset_preserves_libfabric_selection(monkeypatch): ) def test_flag_unset_preserves_legacy_backend_env(monkeypatch, selector, expected_backend): monkeypatch.setenv(selector, "1") - config = CacheTransceiverConfig(backend="DEFAULT") + config = CacheTransceiverConfig(backend="DEFAULT", transceiver_runtime="CPP") constructor = Mock(return_value=object()) monkeypatch.setattr(transceiver_module, "BindKvCacheTransceiver", constructor) @@ -608,7 +608,7 @@ def test_flag_unset_preserves_legacy_backend_env_precedence(monkeypatch): "TRTLLM_USE_NIXL_KVCACHE", ): monkeypatch.setenv(selector, "1") - config = CacheTransceiverConfig(backend="DEFAULT") + config = CacheTransceiverConfig(backend="DEFAULT", transceiver_runtime="CPP") monkeypatch.setattr(transceiver_module, "BindKvCacheTransceiver", Mock()) transceiver_module.create_kv_cache_transceiver(Mock(), Mock(), Mock(), Mock(), config) diff --git a/tests/unittest/_torch/executor/kv_cache/test_kv_cache_budget_split.py b/tests/unittest/_torch/executor/kv_cache/test_kv_cache_budget_split.py index de1a4ae95ddc..0048c2200372 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kv_cache_budget_split.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kv_cache_budget_split.py @@ -64,6 +64,7 @@ def _make_creator( c._max_seq_len = 1024 c._max_num_tokens = 0 c._max_batch_size = 1 + c._max_beam_width = 1 c._is_disagg = False c._cache_transceiver_config = None c._speculative_config = None diff --git a/tests/unittest/_torch/executor/kv_cache/test_mamba_cache_manager.py b/tests/unittest/_torch/executor/kv_cache/test_mamba_cache_manager.py index 1f920c7b718c..a3c86a777a8c 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_mamba_cache_manager.py +++ b/tests/unittest/_torch/executor/kv_cache/test_mamba_cache_manager.py @@ -843,6 +843,7 @@ def test_hybrid_cache_manager_factory_rejects_mixed_override_with_reuse( KvCacheConfig( enable_block_reuse=True, mamba_state_config=MambaStateConfig(periodic_snapshot_interval=256), + use_kv_cache_manager_v2=False, ), ) @@ -901,7 +902,10 @@ def test_hybrid_models_prefer_v2_and_python_transceiver(monkeypatch): ): llm_args = TorchLlmArgs( model="/tmp/dummy_model", - cache_transceiver_config=CacheTransceiverConfig(backend="DEFAULT"), + kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"), + cache_transceiver_config=CacheTransceiverConfig( + backend="DEFAULT", transceiver_runtime="auto" + ), ) _resolve_transceiver_runtime_auto(llm_args, model_cls) _resolve_kv_cache_manager_v2_auto(llm_args, model_cls) @@ -969,6 +973,7 @@ def test_kimi_without_v2_preference_uses_mixed_manager( kv_cache_config=KvCacheConfig( enable_block_reuse=False, tokens_per_block=64, + use_kv_cache_manager_v2="auto", ), ) resolved = _resolve_kv_cache_manager_v2_auto(llm_args, KimiLinearForCausalLM) @@ -995,7 +1000,7 @@ def test_kimi_preferred_transceiver_runtime() -> None: "cache_transceiver_config", [ None, - CacheTransceiverConfig(backend="NIXL"), # runtime left at 'auto' + CacheTransceiverConfig(backend="NIXL", transceiver_runtime="auto"), CacheTransceiverConfig(backend="NIXL", transceiver_runtime="CPP"), CacheTransceiverConfig(backend="UCX", transceiver_runtime="CPP"), CacheTransceiverConfig(backend="UCX", transceiver_runtime="PYTHON"), @@ -1031,7 +1036,10 @@ def test_kimi_disagg_python_nixl_routes_to_mixed_manager( assert ( get_kv_cache_manager_cls( _kimi_model_config(), - KvCacheConfig(enable_block_reuse=False), + KvCacheConfig( + enable_block_reuse=False, + use_kv_cache_manager_v2=False, + ), is_disagg=True, cache_transceiver_config=CacheTransceiverConfig( backend="NIXL", transceiver_runtime="PYTHON" diff --git a/tests/unittest/_torch/modeling/test_modeling_gpt_oss.py b/tests/unittest/_torch/modeling/test_modeling_gpt_oss.py index 197c9897242e..358d9ce0d066 100644 --- a/tests/unittest/_torch/modeling/test_modeling_gpt_oss.py +++ b/tests/unittest/_torch/modeling/test_modeling_gpt_oss.py @@ -63,9 +63,10 @@ def _resolve_gpt_oss_kv_cache_manager_v2(**llm_args_kwargs) -> bool: return _resolve_kv_cache_manager_v2_auto(llm_args, GptOssForCausalLM) -def test_gpt_oss_model_preference_selects_v2(): +def test_gpt_oss_auto_selects_model_preference(): """GPT-OSS is VSWA, so "auto" resolves to KVCacheManagerV2.""" - assert _resolve_gpt_oss_kv_cache_manager_v2() is True + assert _resolve_gpt_oss_kv_cache_manager_v2(kv_cache_config=KvCacheConfig( + use_kv_cache_manager_v2="auto")) is True @pytest.mark.parametrize("user_setting", [False, True]) diff --git a/tests/unittest/_torch/speculative/test_eagle3.py b/tests/unittest/_torch/speculative/test_eagle3.py index 655571c9092e..1c2556646998 100644 --- a/tests/unittest/_torch/speculative/test_eagle3.py +++ b/tests/unittest/_torch/speculative/test_eagle3.py @@ -773,7 +773,8 @@ def test_eagle3_spec_decoding_stats(eagle3_one_model): pytest.skip(f"Required models not found") kv_cache_config = KvCacheConfig(enable_block_reuse=False, - free_gpu_memory_fraction=0.6) + free_gpu_memory_fraction=0.6, + use_kv_cache_manager_v2=eagle3_one_model) spec_config = Eagle3DecodingConfig( max_draft_len=3, speculative_model=eagle_model_dir, @@ -863,8 +864,10 @@ def test_llama_eagle3_long_prompt(use_cuda_graph): else: cuda_graph_config = None + kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2=False) llm_spec = LLM(model=target_model_dir, speculative_config=spec_config, + kv_cache_config=kv_cache_config, max_batch_size=1, cuda_graph_config=cuda_graph_config, disable_overlap_scheduler=True) @@ -878,6 +881,7 @@ def test_llama_eagle3_long_prompt(use_cuda_graph): llm_spec.shutdown() llm_ref = LLM(model=target_model_dir, + kv_cache_config=kv_cache_config, max_batch_size=1, cuda_graph_config=None, disable_overlap_scheduler=False) @@ -1066,7 +1070,8 @@ def test_multi_eagle3(use_one_model: bool): max_batch_size = 16 max_draft_len = 3 kv_cache_config = KvCacheConfig(enable_block_reuse=enable_block_reuse, - free_gpu_memory_fraction=0.5) + free_gpu_memory_fraction=0.5, + use_kv_cache_manager_v2=use_one_model) cuda_graph_config = CudaGraphConfig( batch_sizes=[1]) if use_cuda_graph else None diff --git a/tests/unittest/llmapi/test_async_llm.py b/tests/unittest/llmapi/test_async_llm.py index 9468eedaa6d5..7c9f0b6f9826 100644 --- a/tests/unittest/llmapi/test_async_llm.py +++ b/tests/unittest/llmapi/test_async_llm.py @@ -142,7 +142,7 @@ async def test_async_llm_placement_api(setup_ray_cluster, monkeypatch): @pytest.mark.asyncio async def test_async_llm_reset_prefix_cache(): llama_model_path = str(llm_models_root() / "llama-models-v2/TinyLlama-1.1B-Chat-v1.0") - kv_cache_config = KvCacheConfig(enable_block_reuse=True) + kv_cache_config = KvCacheConfig(enable_block_reuse=True, use_kv_cache_manager_v2=False) prompt = "The future of AI is " * 20 sampling_params = SamplingParams(temperature=0, max_tokens=5, return_perf_metrics=True) diff --git a/tests/unittest/llmapi/test_llm_args.py b/tests/unittest/llmapi/test_llm_args.py index dca3820ed3ee..b1e3a17854f8 100644 --- a/tests/unittest/llmapi/test_llm_args.py +++ b/tests/unittest/llmapi/test_llm_args.py @@ -853,10 +853,8 @@ def test_nvfp4_resolution_preserves_frozen_checkpoint_config( with pytest.raises(AttributeError, match="instance is frozen"): config.attn_backend = "TRTLLM" - @pytest.mark.parametrize("explicit_auto", [False, True]) - def test_auto_uses_model_preference(self, explicit_auto): - kv_cache_config = (KvCacheConfig(use_kv_cache_manager_v2="auto") - if explicit_auto else KvCacheConfig()) + def test_auto_uses_model_preference(self): + kv_cache_config = KvCacheConfig(use_kv_cache_manager_v2="auto") llm_args = TorchLlmArgs(model="/tmp/dummy_model", kv_cache_config=kv_cache_config) @@ -865,7 +863,10 @@ def test_auto_uses_model_preference(self, explicit_auto): assert llm_args.kv_cache_config.use_kv_cache_manager_v2 is True def test_auto_without_preference_falls_back_to_v1(self): - llm_args = TorchLlmArgs(model="/tmp/dummy_model") + llm_args = TorchLlmArgs( + model="/tmp/dummy_model", + kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"), + ) _resolve_kv_cache_manager_v2_auto(llm_args) @@ -882,6 +883,7 @@ def test_auto_without_preference_falls_back_to_v1(self): def test_auto_v2_falls_back_for_incompatible_disagg(self, backend, runtime): llm_args = TorchLlmArgs( model="/tmp/dummy_model", + kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"), cache_transceiver_config=CacheTransceiverConfig( backend=backend, transceiver_runtime=runtime), ) @@ -893,6 +895,7 @@ def test_auto_v2_falls_back_for_incompatible_disagg(self, backend, runtime): def test_auto_v2_keeps_python_nixl_preference(self): llm_args = TorchLlmArgs( model="/tmp/dummy_model", + kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"), cache_transceiver_config=CacheTransceiverConfig( backend="NIXL", transceiver_runtime="PYTHON"), ) @@ -1015,7 +1018,7 @@ def test_KvCacheConfig_declaration(): assert KvCacheConfig().kv_cache_event_hash_algo == "auto" assert KvCacheConfig().block_reuse_config == BlockReuseConfig() assert KvCacheConfig().enable_swa_scratch_reuse is False - assert KvCacheConfig().use_kv_cache_manager_v2 == "auto" + assert KvCacheConfig().use_kv_cache_manager_v2 is True assert KvCacheConfig( use_kv_cache_manager_v2=True).use_kv_cache_manager_v2 is True assert KvCacheConfig( @@ -4375,15 +4378,17 @@ class TestTransceiverRuntimeAutoResolution: """Tests for the transceiver_runtime 'auto' selection mechanism.""" def _disagg_args(self, backend="NIXL", **cfg_kwargs): + cfg_kwargs.setdefault("transceiver_runtime", "auto") return TorchLlmArgs( model="/tmp/dummy_model", + kv_cache_config=KvCacheConfig(use_kv_cache_manager_v2="auto"), cache_transceiver_config=CacheTransceiverConfig(backend=backend, **cfg_kwargs), ) - def test_default_is_auto(self): + def test_default_is_python(self): cfg = CacheTransceiverConfig(backend="NIXL") - assert cfg.transceiver_runtime == "auto" + assert cfg.transceiver_runtime == "PYTHON" def test_auto_no_model_preference_defaults_to_python(self) -> None: """'auto' with no model preference resolves to the Python transceiver.""" @@ -4391,11 +4396,9 @@ def test_auto_no_model_preference_defaults_to_python(self) -> None: _resolve_transceiver_runtime_auto(args) assert args.cache_transceiver_config.transceiver_runtime == "PYTHON" - @pytest.mark.parametrize("explicit_auto", [False, True]) - def test_model_preference_adopted(self, explicit_auto): - """Model preference applies whether 'auto' is implicit or explicit.""" - cfg_kwargs = {"transceiver_runtime": "auto"} if explicit_auto else {} - args = self._disagg_args(**cfg_kwargs) + def test_explicit_auto_adopts_model_preference(self): + """An explicit 'auto' adopts the model preference.""" + args = self._disagg_args() _resolve_transceiver_runtime_auto(args, _PreferPythonTransceiverModel) assert args.cache_transceiver_config.transceiver_runtime == "PYTHON" @@ -4513,7 +4516,7 @@ def test_backend_none_is_noop(self): ) _resolve_transceiver_runtime_auto(args, _PreferPythonTransceiverModel) assert args.cache_transceiver_config.backend is None - assert args.cache_transceiver_config.transceiver_runtime == "auto" + assert args.cache_transceiver_config.transceiver_runtime == "PYTHON" def test_invalid_model_preference_raises(self): @@ -4778,7 +4781,8 @@ def test_deepseek_resolves_auto_to_python_on_nixl(self) -> None: DeepseekV3ForCausalLM args = TorchLlmArgs( model="/tmp/dummy_model", - cache_transceiver_config=CacheTransceiverConfig(backend="NIXL"), + cache_transceiver_config=CacheTransceiverConfig( + backend="NIXL", transceiver_runtime="auto"), ) cfg = self._pretrained_config(["DeepseekV3ForCausalLM"], "deepseek_v3") _resolve_transceiver_runtime_auto(args, DeepseekV3ForCausalLM, cfg) diff --git a/tests/unittest/llmapi/test_llm_pytorch.py b/tests/unittest/llmapi/test_llm_pytorch.py index 5a5d80c99382..e2ed6c006473 100644 --- a/tests/unittest/llmapi/test_llm_pytorch.py +++ b/tests/unittest/llmapi/test_llm_pytorch.py @@ -1108,7 +1108,7 @@ async def test_llm_rpc_get_stats_async(): @pytest.mark.threadleak(enabled=False) @pytest.mark.part0 @skip_ray -@pytest.mark.parametrize("transceiver_runtime", [None, "PYTHON"]) +@pytest.mark.parametrize("transceiver_runtime", ["PYTHON"]) def test_llm_context_only_timed_out(transceiver_runtime): tp_size = 1 use_overlap = False @@ -1121,8 +1121,7 @@ def test_llm_context_only_timed_out(transceiver_runtime): enable_iter_req_stats=enable_iter_req_stats, disable_overlap_scheduler=not use_overlap)) - # Python transceiver (V2) only supports NIXL/DEFAULT backends - backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX" + backend = "NIXL" llm = LLM(model=llama_model_path, kv_cache_config=global_kvcache_config, tensor_parallel_size=tp_size, @@ -1199,15 +1198,11 @@ def test_llm_context_only_timed_out(transceiver_runtime): @pytest.mark.part0 @skip_ray @pytest.mark.parametrize("sender_future_timeout_ms", [100, 1000]) -@pytest.mark.parametrize("backend", ["NIXL", "UCX"]) -@pytest.mark.parametrize("transceiver_runtime", [None, "PYTHON"]) +@pytest.mark.parametrize("backend", ["NIXL"]) +@pytest.mark.parametrize("transceiver_runtime", ["PYTHON"]) def test_llm_context_only_timed_out_kv_cache_exhausted(sender_future_timeout_ms, backend, transceiver_runtime): - # Python transceiver (V2) only supports NIXL/DEFAULT backends - if transceiver_runtime == "PYTHON" and backend == "UCX": - pytest.skip("Python transceiver (V2) does not support UCX backend") - tp_size = 1 use_overlap = False enable_iter_req_stats = False @@ -1292,7 +1287,7 @@ def test_llm_context_only_timed_out_kv_cache_exhausted(sender_future_timeout_ms, @pytest.mark.private_mpi_session @pytest.mark.timeout(600) @pytest.mark.asyncio -@pytest.mark.parametrize("transceiver_runtime", [None, "PYTHON"]) +@pytest.mark.parametrize("transceiver_runtime", ["PYTHON"]) async def test_llm_disagg_gen_cancelled(transceiver_runtime): tp_size = 1 use_overlap = False @@ -1305,8 +1300,7 @@ async def test_llm_disagg_gen_cancelled(transceiver_runtime): enable_iter_req_stats=enable_iter_req_stats, disable_overlap_scheduler=not use_overlap)) - # Python transceiver (V2) only supports NIXL/DEFAULT backends - backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX" + backend = "NIXL" llm_ctx = LLM(model=llama_model_path, kv_cache_config=global_kvcache_config_no_reuse, tensor_parallel_size=tp_size, @@ -1472,7 +1466,7 @@ async def await_and_record(output, req_id: int): @pytest.mark.part0 @skip_ray @pytest.mark.asyncio -@pytest.mark.parametrize("transceiver_runtime", [None, "PYTHON"]) +@pytest.mark.parametrize("transceiver_runtime", ["PYTHON"]) async def test_llm_disagg_streaming_gen_cancelled(transceiver_runtime): tp_size = 1 use_overlap = False @@ -1485,8 +1479,7 @@ async def test_llm_disagg_streaming_gen_cancelled(transceiver_runtime): enable_iter_req_stats=enable_iter_req_stats, disable_overlap_scheduler=not use_overlap)) - # Python transceiver (V2) only supports NIXL/DEFAULT backends - backend = "NIXL" if transceiver_runtime == "PYTHON" else "UCX" + backend = "NIXL" llm_ctx = LLM(model=llama_model_path, kv_cache_config=global_kvcache_config_no_reuse, tensor_parallel_size=tp_size, diff --git a/tests/unittest/llmapi/test_quickstart_advanced.py b/tests/unittest/llmapi/test_quickstart_advanced.py index d9e9212721c2..3e33e93d0aec 100644 --- a/tests/unittest/llmapi/test_quickstart_advanced.py +++ b/tests/unittest/llmapi/test_quickstart_advanced.py @@ -44,9 +44,9 @@ def test_use_kv_cache_manager_v2_cli_values(cli_value: str, expected: str | bool assert args.use_kv_cache_manager_v2 == expected -def test_use_kv_cache_manager_v2_cli_default_is_auto() -> None: +def test_use_kv_cache_manager_v2_cli_default_is_true() -> None: parser = _MODULE.add_llm_args(argparse.ArgumentParser()) args = parser.parse_args(["--model_dir", "dummy-model"]) - assert args.use_kv_cache_manager_v2 == "auto" + assert args.use_kv_cache_manager_v2 is True diff --git a/tests/unittest/others/test_cache_transceiver_precheck_config.py b/tests/unittest/others/test_cache_transceiver_precheck_config.py index 561d3107a05a..e8c8f379ce08 100644 --- a/tests/unittest/others/test_cache_transceiver_precheck_config.py +++ b/tests/unittest/others/test_cache_transceiver_precheck_config.py @@ -289,9 +289,9 @@ def test_use_kv_cache_manager_v2_flags(): # Absent -> "auto" (the driver resolves it against the model's # get_preferred_kv_cache_manager_version at runtime, like serving). plan = pcfg.resolve_plan(_disagg_yaml()) - assert plan["ctx_use_kv_cache_manager_v2"] == "auto" - assert plan["gen_use_kv_cache_manager_v2"] == "auto" - assert pcfg.side_plan(plan, "ctx")["use_kv_cache_manager_v2"] == "auto" + assert plan["ctx_use_kv_cache_manager_v2"] is True + assert plan["gen_use_kv_cache_manager_v2"] is True + assert pcfg.side_plan(plan, "ctx")["use_kv_cache_manager_v2"] is True # Explicit yaml values win, per side. plan = pcfg.resolve_plan( diff --git a/tests/unittest/others/test_kv_cache_transceiver.py b/tests/unittest/others/test_kv_cache_transceiver.py index 1179365af005..7193a737fc0a 100644 --- a/tests/unittest/others/test_kv_cache_transceiver.py +++ b/tests/unittest/others/test_kv_cache_transceiver.py @@ -415,6 +415,7 @@ def test_cancel_request_in_transmission(attention_type): kv_cache_manager_gen = create_kv_cache_manager(mapping, gen_kv_cache_dtype) cache_transceiver_config = CacheTransceiverConfig(backend="DEFAULT", + transceiver_runtime="CPP", max_tokens_in_buffer=512) kv_cache_transceiver_ctx = create_kv_cache_transceiver( @@ -486,6 +487,7 @@ def test_async_transfer_keeps_llm_request_alive(): kv_cache_manager_gen = create_kv_cache_manager(mapping, DataType.HALF) cache_transceiver_config = CacheTransceiverConfig(backend="DEFAULT", + transceiver_runtime="CPP", max_tokens_in_buffer=512) transceiver_ctx = create_kv_cache_transceiver(mapping, dist, kv_cache_manager_ctx, @@ -612,7 +614,10 @@ def test_kv_transfer_timeout_warns_once_per_request(capfd): kv_cache_manager_ctx = create_kv_cache_manager(mapping, DataType.HALF) cache_transceiver_config = CacheTransceiverConfig( - backend="DEFAULT", max_tokens_in_buffer=512, kv_transfer_timeout_ms=100) + backend="DEFAULT", + transceiver_runtime="CPP", + max_tokens_in_buffer=512, + kv_transfer_timeout_ms=100) transceiver_ctx = create_kv_cache_transceiver(mapping, dist, kv_cache_manager_ctx, AttentionTypeCpp.DEFAULT, @@ -661,6 +666,7 @@ def test_kv_transfer_timeout_silent_when_unset(capfd): kv_cache_manager_ctx = create_kv_cache_manager(mapping, DataType.HALF) cache_transceiver_config = CacheTransceiverConfig(backend="DEFAULT", + transceiver_runtime="CPP", max_tokens_in_buffer=512) transceiver_ctx = create_kv_cache_transceiver(mapping, dist, kv_cache_manager_ctx, @@ -698,6 +704,7 @@ def test_context_transfer_bounded_poll_keeps_request_in_progress(capfd): cache_transceiver_config = CacheTransceiverConfig( backend="DEFAULT", + transceiver_runtime="CPP", max_tokens_in_buffer=512, kv_transfer_timeout_ms=100, kv_transfer_sender_future_timeout_ms=10)