From 209b7b9a4b86329766c712b7d3383e6352c64469 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 10:25:21 +0000 Subject: [PATCH 01/15] [None][fix] Respect KVCM V2 initialization and warmup budgets Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../kv_cache/kv_cache_manager_v2.py | 17 ++- .../_torch/pyexecutor/model_engine.py | 33 +++-- .../defs/accuracy/test_llm_api_pytorch.py | 3 +- .../test_llm_api_pytorch_multimodal.py | 2 +- .../kv_cache/test_kvcm2_integration.py | 129 ++++++++++++++++++ tests/unittest/grpc/smg/test_smg.py | 2 +- 6 files changed, 173 insertions(+), 13 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py index ed0cc2a89afe..9a9cd0c824c6 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py @@ -2817,12 +2817,25 @@ def _build_base_config( min_decode_capacity = 1 + self.max_draft_len + self.num_extra_kv_tokens # Model one request at max_seq_len plus minimal decode requests # to fill constraint_batch_size. + generation_capacity = self.max_seq_len + if type(self) is KVCacheManagerV2 and all( + window is None for window in self.max_attention_window_vec + ): + # Full attention has one lifecycle, so its pool already receives + # the available quota. A max_seq_len floor would silently grow + # that quota when the model's context limit cannot fit in memory. + # Keep the minimum batch floor; CUDA graph warmup queries the + # allocated pool after reserving its short decode requests to + # determine how long the remaining request can actually be. + generation_capacity = min_decode_capacity + # Other layouts need the full-length constraint to distribute + # capacity across their distinct attention/recurrent pools. constraints.append( BatchDesc( [ KVCacheDesc( - capacity=self.max_seq_len, - history_length=self.max_seq_len - 1, + capacity=generation_capacity, + history_length=generation_capacity - 1, ) ] + [KVCacheDesc(capacity=min_decode_capacity, history_length=0)] diff --git a/tensorrt_llm/_torch/pyexecutor/model_engine.py b/tensorrt_llm/_torch/pyexecutor/model_engine.py index 2429877c9cb4..1d354499a023 100644 --- a/tensorrt_llm/_torch/pyexecutor/model_engine.py +++ b/tensorrt_llm/_torch/pyexecutor/model_engine.py @@ -3478,19 +3478,36 @@ def free_warmup_requests() -> None: # Use max_draft_loop_tokens for capacity estimation to account # for the actual KV reservation per request. _kv_draft = self.max_draft_loop_tokens - available_tokens = kv_cache_manager.get_num_available_tokens( - token_num_upper_bound=max_seq_len, - batch_size=batch_size, - max_num_draft_tokens=_kv_draft) - # Also consider draft KV cache capacity when it exists - if draft_kv_cache_manager is not None: - draft_available_tokens = draft_kv_cache_manager.get_num_available_tokens( - batch_size=batch_size, + def get_available_tokens(manager): + capacity_batch_size = batch_size + if type(manager) is KVCacheManagerV2 and all( + window is None + for window in manager.max_attention_window_vec): + # The capacity query reserves one page for each other sequence. + # Full attention has one lifecycle, so use the actual occupied + # pages, including multi-page draft dummies and the guard page. + capacity_batch_size = 1 + sum( + int(cache.num_blocks) + for cache in manager.kv_cache_map.values()) + return manager.get_num_available_tokens( token_num_upper_bound=max_seq_len, + batch_size=capacity_batch_size, max_num_draft_tokens=_kv_draft) + + available_tokens = get_available_tokens(kv_cache_manager) + + # Also consider draft KV cache capacity when it exists + if draft_kv_cache_manager is not None: + draft_available_tokens = get_available_tokens( + draft_kv_cache_manager) available_tokens = min(available_tokens, draft_available_tokens) + if isinstance(kv_cache_manager, KVCacheManagerV2): + # V2's generation dummy reserves one more token after allocating + # its input. Leave room for that token in both target and draft KV. + available_tokens -= 1 + token_num = max( ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1, min( diff --git a/tests/integration/defs/accuracy/test_llm_api_pytorch.py b/tests/integration/defs/accuracy/test_llm_api_pytorch.py index d71b2c35e233..99667216811d 100644 --- a/tests/integration/defs/accuracy/test_llm_api_pytorch.py +++ b/tests/integration/defs/accuracy/test_llm_api_pytorch.py @@ -6126,7 +6126,8 @@ class TestSeedOss_36B(LlmapiAccuracyTestHarness): @pytest.mark.timeout(14400) @pytest.mark.skip_less_device_memory(140000) def test_auto_dtype(self): - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.8) + kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.8, + use_kv_cache_manager_v2=True) chat_template_kwargs = dict(thinking_budget=-1) with LLM(self.MODEL_PATH, kv_cache_config=kv_cache_config) as llm: diff --git a/tests/integration/defs/accuracy/test_llm_api_pytorch_multimodal.py b/tests/integration/defs/accuracy/test_llm_api_pytorch_multimodal.py index 15f437d4f8b0..02591600143d 100644 --- a/tests/integration/defs/accuracy/test_llm_api_pytorch_multimodal.py +++ b/tests/integration/defs/accuracy/test_llm_api_pytorch_multimodal.py @@ -639,7 +639,7 @@ class TestMistralSmall24B(LlmapiAccuracyTestHarness): ids=["forced_chunked_prefill"], ) def test_auto_dtype(self, max_num_tokens): - kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.75) + kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.75, use_kv_cache_manager_v2=True) with LLM( self.MODEL_PATH, kv_cache_config=kv_cache_config, diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index f7d26f03a0b0..26fa6c9761b3 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -36,6 +36,8 @@ _update_kv_cache_draft_token_location, ) from tensorrt_llm._torch.pyexecutor.llm_request import LlmRequest, LlmRequestState +from tensorrt_llm._torch.pyexecutor.model_engine import PyTorchModelEngine +from tensorrt_llm._torch.pyexecutor.resource_manager import ResourceManager, ResourceManagerType from tensorrt_llm._torch.pyexecutor.scheduler import ScheduledRequests from tensorrt_llm.bindings import DataType, SamplingConfig from tensorrt_llm.bindings.BuildInfo import ENABLE_MULTI_DEVICE @@ -895,6 +897,7 @@ def test_avg_seq_len_builds_warmup_constraints() -> None: max_seq_len=1024, max_num_tokens=2048, max_draft_len=2, + max_attention_window_vec=[None, 256], ) assert config.typical_step == BatchDesc( @@ -913,6 +916,132 @@ def test_avg_seq_len_builds_warmup_constraints() -> None: ] +@pytest.fixture(params=[0, 16], ids=["decode", "speculative"]) +def _budget_warmup_spec_config(request: pytest.FixtureRequest): + return ( + Eagle3DecodingConfig(max_draft_len=request.param, speculative_model="dummy") + if request.param + else None + ) + + +@pytest.fixture(params=[False, True], ids=["no_guard", "guard"]) +def _budget_warmup_guard_page(request: pytest.FixtureRequest, monkeypatch): + if request.param: + monkeypatch.setenv("TRTLLM_KV_GUARD_PAGE", "1") + else: + monkeypatch.delenv("TRTLLM_KV_GUARD_PAGE", raising=False) + return request.param + + +@pytest.fixture(params=["bytes", "tokens"]) +def _budget_limited_full_attention_manager( + request: pytest.FixtureRequest, + _budget_warmup_spec_config, + _budget_warmup_guard_page, +): + if not torch.cuda.is_available(): + pytest.skip("requires CUDA") + init_cuda_once() + budget = {"max_gpu_total_bytes": 17 << 20} if request.param == "bytes" else {"max_tokens": 8192} + config = KvCacheConfig( + use_kv_cache_manager_v2=True, + enable_block_reuse=False, + host_cache_size=0, + avg_seq_len=32768, + **budget, + ) + manager = KVCacheManagerV2( + config, + CacheType.SELF, + num_layers=2, + num_kv_heads=2, + head_dim=128, + tokens_per_block=32, + max_seq_len=131072, + max_batch_size=8, + max_num_tokens=2048, + mapping=Mapping(), + dtype=DataType.HALF, + spec_config=_budget_warmup_spec_config, + ) + try: + assert len(manager.kv_cache_map) == int(_budget_warmup_guard_page) + yield manager + finally: + manager.shutdown() + + +def test_full_attention_warmup_respects_allocated_budget( + _budget_limited_full_attention_manager: KVCacheManagerV2, +) -> None: + manager = _budget_limited_full_attention_manager + requested_quota = manager.kv_cache_manager_py_config.cache_tiers[0].quota + allocated_bytes = manager.impl.get_quota(kv_cache_v2_module.GPU_LEVEL) + # These small quotas round up to a 2 MiB GPU allocation grain. The model's + # full context would require 256 MiB, far beyond either configured budget. + assert 0 < allocated_bytes <= requested_quota + (2 << 20) + assert manager.max_num_tokens < manager.max_seq_len < 131072 + + requests = manager.add_dummy_requests( + [0], token_nums=[manager.max_num_tokens // 2], is_gen=False + ) + assert requests is not None + try: + cache = manager.kv_cache_map[requests[0].py_request_id] + assert cache.resize(manager.max_num_tokens, history_length=0) + assert cache.capacity == manager.max_num_tokens + cache.suspend() + assert cache.resume(torch.cuda.current_stream().cuda_stream) + finally: + for request in requests: + manager.free_resources(request) + + +def test_full_attention_budget_supports_cuda_graph_warmup( + _budget_limited_full_attention_manager: KVCacheManagerV2, + _budget_warmup_spec_config, +) -> None: + manager = _budget_limited_full_attention_manager + # Exercise the real graph request builder: it allocates the short requests, + # queries the remaining capacity, then grows the longest generation request. + engine = SimpleNamespace() + engine.kv_cache_manager_key = ResourceManagerType.KV_CACHE_MANAGER + engine.spec_config = _budget_warmup_spec_config + engine.max_beam_width = 1 + engine.max_draft_loop_tokens = manager.max_draft_len + engine.max_seq_len = 131072 + engine.use_mrope = False + engine.get_runtime_tokens_per_gen_step = lambda draft_len: draft_len + 1 + engine._get_draft_kv_cache_manager = lambda _: None + engine._is_encoder_decoder_model = lambda: False + engine.model = SimpleNamespace( + model_config=SimpleNamespace(pretrained_config=SimpleNamespace()) + ) + resources = ResourceManager({ResourceManagerType.KV_CACHE_MANAGER: manager}) + + batch = PyTorchModelEngine._create_cuda_graph_warmup_request( + engine, resources, batch_size=manager.max_batch_size, draft_len=manager.max_draft_len + ) + assert batch is not None + requests = list(batch.generation_requests) + try: + assert len(requests) == manager.max_batch_size + longest_request = requests[0] + cache = manager.kv_cache_map[longest_request.py_request_id] + assert cache.capacity > manager.max_num_tokens + manager.free_resources(longest_request) + requests.remove(longest_request) + # Once the longest request finishes, another generation request can + # grow across page boundaries into the released capacity. + cache = manager.kv_cache_map[requests[0].py_request_id] + assert cache.capacity < manager.max_num_tokens + assert cache.resize(manager.max_num_tokens, history_length=cache.history_length + 1) + finally: + for request in requests: + manager.free_resources(request) + + def test_avg_seq_len_updates_typical_step() -> None: config = _make_cache_config_for_test( KvCacheConfig(avg_seq_len=256), diff --git a/tests/unittest/grpc/smg/test_smg.py b/tests/unittest/grpc/smg/test_smg.py index bbf6f5a31717..8cd1909517ef 100644 --- a/tests/unittest/grpc/smg/test_smg.py +++ b/tests/unittest/grpc/smg/test_smg.py @@ -842,7 +842,7 @@ def grpc_vlm_service(): model_path = get_model_path(vlm_model_name) llm = LLM( model=model_path, - kv_cache_config=KvCacheConfig(free_gpu_memory_fraction=0.6), + kv_cache_config=KvCacheConfig(free_gpu_memory_fraction=0.6, use_kv_cache_manager_v2=True), load_format="dummy", ) tokenizer = llm.tokenizer From b1d7f2b7604da7d54e06cb897fff0b0d98b85a5e Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 10:26:40 +0000 Subject: [PATCH 02/15] [None][test] Preserve full and mixed attention constraint coverage Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../executor/kv_cache/test_kvcm2_integration.py | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index 26fa6c9761b3..a5133fe54f68 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -890,14 +890,19 @@ def test_default_uses_allocator_fallback() -> None: assert config.constraints == [] -def test_avg_seq_len_builds_warmup_constraints() -> None: +@pytest.mark.parametrize( + "attention_windows,generation_capacity", + [([None, None], 3), ([None, 256], 1024)], + ids=["full_attention", "mixed_attention"], +) +def test_avg_seq_len_builds_warmup_constraints(attention_windows, generation_capacity) -> None: config = _make_cache_config_for_test( - KvCacheConfig(host_cache_size=0, avg_seq_len=1024), + KvCacheConfig(use_kv_cache_manager_v2=True, host_cache_size=0, avg_seq_len=1024), max_batch_size=3, max_seq_len=1024, max_num_tokens=2048, max_draft_len=2, - max_attention_window_vec=[None, 256], + max_attention_window_vec=attention_windows, ) assert config.typical_step == BatchDesc( @@ -907,7 +912,7 @@ def test_avg_seq_len_builds_warmup_constraints() -> None: assert config.constraints == [ BatchDesc( [ - KVCacheDesc(capacity=1024, history_length=1023), + KVCacheDesc(capacity=generation_capacity, history_length=generation_capacity - 1), KVCacheDesc(capacity=3, history_length=0), KVCacheDesc(capacity=3, history_length=0), ] @@ -979,7 +984,7 @@ def test_full_attention_warmup_respects_allocated_budget( requested_quota = manager.kv_cache_manager_py_config.cache_tiers[0].quota allocated_bytes = manager.impl.get_quota(kv_cache_v2_module.GPU_LEVEL) # These small quotas round up to a 2 MiB GPU allocation grain. The model's - # full context would require 256 MiB, far beyond either configured budget. + # full context requires at least 256 MiB, far beyond either configured budget. assert 0 < allocated_bytes <= requested_quota + (2 << 20) assert manager.max_num_tokens < manager.max_seq_len < 131072 From 26033aacac840c789daa0296119e0e32f16c3670 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 10:57:59 +0000 Subject: [PATCH 03/15] [None][fix] Limit initialization floor changes to self attention Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../pyexecutor/kv_cache/kv_cache_manager_v2.py | 12 +++++++----- .../executor/kv_cache/test_kvcm2_integration.py | 16 ++++++++++++---- 2 files changed, 19 insertions(+), 9 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py index 9a9cd0c824c6..2550868f95c3 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py @@ -2818,18 +2818,20 @@ def _build_base_config( # Model one request at max_seq_len plus minimal decode requests # to fill constraint_batch_size. generation_capacity = self.max_seq_len - if type(self) is KVCacheManagerV2 and all( - window is None for window in self.max_attention_window_vec + if ( + type(self) is KVCacheManagerV2 + and self.kv_cache_type == CacheTypeCpp.SELF + and all(window is None for window in self.max_attention_window_vec) ): - # Full attention has one lifecycle, so its pool already receives + # Full self attention has one lifecycle, so its pool receives # the available quota. A max_seq_len floor would silently grow # that quota when the model's context limit cannot fit in memory. # Keep the minimum batch floor; CUDA graph warmup queries the # allocated pool after reserving its short decode requests to # determine how long the remaining request can actually be. generation_capacity = min_decode_capacity - # Other layouts need the full-length constraint to distribute - # capacity across their distinct attention/recurrent pools. + # Preserve the full-length constraint for other cache types and + # layouts. Cross attention sizes encoder warmup independently. constraints.append( BatchDesc( [ diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index a5133fe54f68..5ae8cc3f4905 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -891,13 +891,21 @@ def test_default_uses_allocator_fallback() -> None: @pytest.mark.parametrize( - "attention_windows,generation_capacity", - [([None, None], 3), ([None, 256], 1024)], - ids=["full_attention", "mixed_attention"], + "kv_cache_type,attention_windows,generation_capacity", + [ + (CacheType.SELF, [None, None], 3), + (CacheType.SELF, [None, 256], 1024), + (CacheType.CROSS, [None, None], 1024), + (CacheType.SELFKONLY, [None, None], 1024), + ], + ids=["full_attention", "mixed_attention", "cross_attention", "key_only"], ) -def test_avg_seq_len_builds_warmup_constraints(attention_windows, generation_capacity) -> None: +def test_avg_seq_len_builds_warmup_constraints( + kv_cache_type, attention_windows, generation_capacity +) -> None: config = _make_cache_config_for_test( KvCacheConfig(use_kv_cache_manager_v2=True, host_cache_size=0, avg_seq_len=1024), + kv_cache_type=kv_cache_type, max_batch_size=3, max_seq_len=1024, max_num_tokens=2048, From 8aee22812bf37e4919ab387c743fd495e1ec9572 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 15 Sep 2026 07:30:19 -0700 Subject: [PATCH 04/15] [None][test] Cover multimodal examples with KVCM V2 budget fix Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- tests/unittest/llmapi/apps/_test_openai_chat_multimodal.py | 4 ++++ .../llmapi/apps/_test_trtllm_serve_multimodal_example.py | 1 + 2 files changed, 5 insertions(+) diff --git a/tests/unittest/llmapi/apps/_test_openai_chat_multimodal.py b/tests/unittest/llmapi/apps/_test_openai_chat_multimodal.py index 03019f68f0fd..b45d2b9d1716 100644 --- a/tests/unittest/llmapi/apps/_test_openai_chat_multimodal.py +++ b/tests/unittest/llmapi/apps/_test_openai_chat_multimodal.py @@ -1,3 +1,6 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + import io import os import tempfile @@ -43,6 +46,7 @@ def temp_extra_llm_api_options_file(request): "kv_cache_config": { "enable_block_reuse": False, "free_gpu_memory_fraction": 0.6, + "use_kv_cache_manager_v2": True, }, } diff --git a/tests/unittest/llmapi/apps/_test_trtllm_serve_multimodal_example.py b/tests/unittest/llmapi/apps/_test_trtllm_serve_multimodal_example.py index de7aa43fa301..c5a03e59ef17 100644 --- a/tests/unittest/llmapi/apps/_test_trtllm_serve_multimodal_example.py +++ b/tests/unittest/llmapi/apps/_test_trtllm_serve_multimodal_example.py @@ -40,6 +40,7 @@ def temp_extra_llm_api_options_file(request): "kv_cache_config": { "enable_block_reuse": False, "free_gpu_memory_fraction": 0.6, + "use_kv_cache_manager_v2": True, }, "max_num_tokens": 16384, # for pytorch backend # NOTE: This is for video support. From 06a47f04adf52a70552345a82e52e2b45f423047 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Wed, 16 Sep 2026 05:45:53 +0000 Subject: [PATCH 05/15] [None][fix] Estimate V2 cache constraints and query warmup capacity Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../kv_cache/kv_cache_manager_v2.py | 140 ++++++++++++------ .../_torch/pyexecutor/model_engine.py | 67 +++++---- .../kv_cache/test_kvcm2_integration.py | 40 +++-- 3 files changed, 157 insertions(+), 90 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py index 2550868f95c3..84ae6d067c46 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py @@ -2798,51 +2798,34 @@ def _build_base_config( * (generation_request_capacity - 1) ) - # CUDA graph generation warmup uses one request at max_seq_len and - # enough minimal decode requests to fill the resident capacity. - if ( - self.max_cuda_graph_batch_size is not None - and self.max_cuda_graph_batch_size > 0 - and self.is_estimating_kv_cache - and all(window is None for window in self.max_attention_window_vec) - ): - # Estimation graph warmup needs the smaller of the resident - # capacity and the largest captured CUDA graph batch. - constraint_batch_size = min( - generation_request_capacity, self.max_cuda_graph_batch_size - ) - else: - constraint_batch_size = generation_request_capacity - constraint_batch_size = max(1, constraint_batch_size) min_decode_capacity = 1 + self.max_draft_len + self.num_extra_kv_tokens - # Model one request at max_seq_len plus minimal decode requests - # to fill constraint_batch_size. - generation_capacity = self.max_seq_len - if ( - type(self) is KVCacheManagerV2 - and self.kv_cache_type == CacheTypeCpp.SELF - and all(window is None for window in self.max_attention_window_vec) - ): - # Full self attention has one lifecycle, so its pool receives - # the available quota. A max_seq_len floor would silently grow - # that quota when the model's context limit cannot fit in memory. - # Keep the minimum batch floor; CUDA graph warmup queries the - # allocated pool after reserving its short decode requests to - # determine how long the remaining request can actually be. - generation_capacity = min_decode_capacity - # Preserve the full-length constraint for other cache types and - # layouts. Cross attention sizes encoder warmup independently. - constraints.append( - BatchDesc( - [ - KVCacheDesc( - capacity=generation_capacity, - history_length=generation_capacity - 1, - ) - ] - + [KVCacheDesc(capacity=min_decode_capacity, history_length=0)] - * (constraint_batch_size - 1) - ) + gpu_quota = next( + tier.quota for tier in cache_tiers if isinstance(tier, GpuCacheTierConfig) + ) + # Native minimum slot counts are divided by the resume watermark. + # Normalize the quota before estimating a feasible long request. + estimate = self._get_max_tokens_from_quota( + int(gpu_quota * kv_cache_config.max_util_for_resume) + ) + generation_capacity = int(min(self.max_seq_len, max(min_decode_capacity, estimate))) + # These are independent workloads. Graph warmup shortens its long + # request after allocating the short requests; requiring both at + # this estimated length would count their memory twice. + constraints.extend( + [ + BatchDesc( + [ + KVCacheDesc( + capacity=generation_capacity, + history_length=generation_capacity - 1, + ) + ] + ), + BatchDesc( + [KVCacheDesc(capacity=min_decode_capacity, history_length=0)] + * self.max_batch_size + ), + ] ) # General and chunked-prefill warmup uses one fresh context request @@ -3251,6 +3234,75 @@ def get_num_available_tokens( clamped = min(clamped, self._gpu_max_tokens - extra_tokens) return clamped + def get_warmup_token_capacity( + self, + *, + token_num_upper_bound: int, + max_num_draft_tokens: int = 0, + draft_kv_cache_manager: Optional["KVCacheManagerV2"] = None, + ) -> int: + """Return an input length that fits one additional generation dummy. + + Other dummies must already be resident. This query is for exclusive + warmup, not concurrent request admission, and never grows the pool. + The descriptors match both resize calls in ``add_dummy_requests``: + the target keeps generation history, while a coupled draft cache + materializes the entire input. All sequence/position limits must be + applied to the upper bound before calling this method. + """ + managers = [self] + if draft_kv_cache_manager is not None: + managers.append(draft_kv_cache_manager) + available_slots = [] + overhead = self.num_extra_kv_tokens + max_num_draft_tokens + 1 + upper = token_num_upper_bound + for manager in managers: + statistics = manager._get_storage_statistics(GPU_LEVEL) + max_util = manager.kv_cache_manager_py_config.max_util_for_resume + if any(stat.total and stat.unavailable / stat.total > max_util for stat in statistics): + return 0 + available_slots.append([stat.available for stat in statistics]) + if manager._gpu_max_tokens is not None: + upper = min(upper, manager._gpu_max_tokens - overhead) + minimum_tokens = 2 if self._has_cp_helix else 1 + if upper < minimum_tokens: + return 0 + + def fits(tokens: int) -> bool: + for index, (manager, available) in enumerate(zip(managers, available_slots)): + materialize_history = index != 0 + descriptor = KVCacheDesc( + capacity=tokens + overhead, + history_length=0 if materialize_history else tokens - 1, + ) + needed = _introspection.compute_slots_for_batch( + manager.impl, + BatchDesc([descriptor]), + manager._ledger_tokens_per_block, + manager.kv_cache_manager_py_config.swa_scratch_reuse + if materialize_history + else None, + ) + if any(required > free for required, free in zip(needed, available)): + return False + return True + + if fits(upper): + return upper + if not fits(minimum_tokens): + return 0 + # SWA retention can oscillate by one slot within a page. Search one + # common page phase, checking target and draft at the same input length. + page = math.lcm(*(manager._ledger_tokens_per_block for manager in managers)) + lo, hi = 0, upper // page + 1 + while hi - lo > 1: + mid = (lo + hi) // 2 + if fits(mid * page): + lo = mid + else: + hi = mid + return max(minimum_tokens, lo * page) + def get_num_free_blocks(self) -> int: # NOTE This method is used to get the number of blocks in the primary pool not the FREE blocks. # However, since we only use this function when the kv cache manager is empty, so it is safe to do so. diff --git a/tensorrt_llm/_torch/pyexecutor/model_engine.py b/tensorrt_llm/_torch/pyexecutor/model_engine.py index 1d354499a023..06c348b44ebd 100644 --- a/tensorrt_llm/_torch/pyexecutor/model_engine.py +++ b/tensorrt_llm/_torch/pyexecutor/model_engine.py @@ -3479,40 +3479,10 @@ def free_warmup_requests() -> None: # for the actual KV reservation per request. _kv_draft = self.max_draft_loop_tokens - def get_available_tokens(manager): - capacity_batch_size = batch_size - if type(manager) is KVCacheManagerV2 and all( - window is None - for window in manager.max_attention_window_vec): - # The capacity query reserves one page for each other sequence. - # Full attention has one lifecycle, so use the actual occupied - # pages, including multi-page draft dummies and the guard page. - capacity_batch_size = 1 + sum( - int(cache.num_blocks) - for cache in manager.kv_cache_map.values()) - return manager.get_num_available_tokens( - token_num_upper_bound=max_seq_len, - batch_size=capacity_batch_size, - max_num_draft_tokens=_kv_draft) - - available_tokens = get_available_tokens(kv_cache_manager) - - # Also consider draft KV cache capacity when it exists - if draft_kv_cache_manager is not None: - draft_available_tokens = get_available_tokens( - draft_kv_cache_manager) - available_tokens = min(available_tokens, draft_available_tokens) - - if isinstance(kv_cache_manager, KVCacheManagerV2): - # V2's generation dummy reserves one more token after allocating - # its input. Leave room for that token in both target and draft KV. - available_tokens -= 1 - - token_num = max( - ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1, - min( - available_tokens, max_seq_len - 1 - - get_num_extra_kv_tokens(self.spec_config) - _kv_draft)) + # Apply every static limit before the V2 query. Changing its result + # afterward can increase SWA page demand at a different page phase. + token_num = max_seq_len - 1 - get_num_extra_kv_tokens( + self.spec_config) - _kv_draft model_config = self.model.model_config.pretrained_config max_position_embeddings = getattr(model_config, 'max_position_embeddings', None) @@ -3529,6 +3499,35 @@ def get_available_tokens(manager): if max_position_embeddings is not None: token_num = min(token_num, max_position_embeddings - _kv_draft) + if isinstance(kv_cache_manager, KVCacheManagerV2): + assert draft_kv_cache_manager is None or isinstance( + draft_kv_cache_manager, KVCacheManagerV2) + token_num = kv_cache_manager.get_warmup_token_capacity( + token_num_upper_bound=token_num, + max_num_draft_tokens=_kv_draft, + draft_kv_cache_manager=draft_kv_cache_manager) + minimum_tokens = ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1 + if token_num < minimum_tokens: + free_warmup_requests() + return None + else: + available_tokens = kv_cache_manager.get_num_available_tokens( + token_num_upper_bound=max_seq_len, + batch_size=batch_size, + max_num_draft_tokens=_kv_draft) + if draft_kv_cache_manager is not None: + available_tokens = min( + available_tokens, + draft_kv_cache_manager.get_num_available_tokens( + token_num_upper_bound=max_seq_len, + batch_size=batch_size, + max_num_draft_tokens=_kv_draft)) + token_num = max( + ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1, + min(token_num, available_tokens)) + if max_position_embeddings is not None: + token_num = min(token_num, max_position_embeddings - _kv_draft) + token_num = int( token_num) # Ensure int for range() in add_dummy_requests diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index 5ae8cc3f4905..b2f0d0baed28 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -218,6 +218,7 @@ def _make_cache_config_for_test( cache_manager.enable_joint_kv_cache_reuse = False cache_manager.reuse_match_backoff = 0 cache_manager.get_layer_bytes_per_token = lambda **_: 128 + cache_manager._get_max_tokens_from_quota = lambda _: max_seq_len # Mirrors __init__: without helix the ledger block equals the physical # page (the helper re-enacts construction for partial instances). cache_manager._ledger_tokens_per_block = 128 @@ -893,7 +894,7 @@ def test_default_uses_allocator_fallback() -> None: @pytest.mark.parametrize( "kv_cache_type,attention_windows,generation_capacity", [ - (CacheType.SELF, [None, None], 3), + (CacheType.SELF, [None, None], 1024), (CacheType.SELF, [None, 256], 1024), (CacheType.CROSS, [None, None], 1024), (CacheType.SELFKONLY, [None, None], 1024), @@ -921,10 +922,9 @@ def test_avg_seq_len_builds_warmup_constraints( BatchDesc( [ KVCacheDesc(capacity=generation_capacity, history_length=generation_capacity - 1), - KVCacheDesc(capacity=3, history_length=0), - KVCacheDesc(capacity=3, history_length=0), ] ), + BatchDesc([KVCacheDesc(capacity=3, history_length=0)] * 3), BatchDesc([KVCacheDesc(capacity=2048, history_length=0)]), ] @@ -947,11 +947,23 @@ def _budget_warmup_guard_page(request: pytest.FixtureRequest, monkeypatch): return request.param +@pytest.fixture(params=[False, True], ids=["final", "estimation"]) +def _budget_warmup_phase(request: pytest.FixtureRequest): + return request.param + + +@pytest.fixture(params=[None, [256, 256], [256, 131072]], ids=["full", "swa", "mixed"]) +def _budget_warmup_windows(request: pytest.FixtureRequest): + return request.param + + @pytest.fixture(params=["bytes", "tokens"]) -def _budget_limited_full_attention_manager( +def _budget_limited_attention_manager( request: pytest.FixtureRequest, _budget_warmup_spec_config, _budget_warmup_guard_page, + _budget_warmup_phase, + _budget_warmup_windows, ): if not torch.cuda.is_available(): pytest.skip("requires CUDA") @@ -962,6 +974,7 @@ def _budget_limited_full_attention_manager( enable_block_reuse=False, host_cache_size=0, avg_seq_len=32768, + max_attention_window=_budget_warmup_windows, **budget, ) manager = KVCacheManagerV2( @@ -977,6 +990,7 @@ def _budget_limited_full_attention_manager( mapping=Mapping(), dtype=DataType.HALF, spec_config=_budget_warmup_spec_config, + is_estimating_kv_cache=_budget_warmup_phase, ) try: assert len(manager.kv_cache_map) == int(_budget_warmup_guard_page) @@ -985,16 +999,18 @@ def _budget_limited_full_attention_manager( manager.shutdown() -def test_full_attention_warmup_respects_allocated_budget( - _budget_limited_full_attention_manager: KVCacheManagerV2, +def test_attention_warmup_respects_allocated_budget( + _budget_limited_attention_manager: KVCacheManagerV2, ) -> None: - manager = _budget_limited_full_attention_manager + manager = _budget_limited_attention_manager requested_quota = manager.kv_cache_manager_py_config.cache_tiers[0].quota allocated_bytes = manager.impl.get_quota(kv_cache_v2_module.GPU_LEVEL) # These small quotas round up to a 2 MiB GPU allocation grain. The model's # full context requires at least 256 MiB, far beyond either configured budget. assert 0 < allocated_bytes <= requested_quota + (2 << 20) - assert manager.max_num_tokens < manager.max_seq_len < 131072 + assert manager.max_num_tokens < manager.max_seq_len <= 131072 + if any(window is None for window in manager.max_attention_window_vec): + assert manager.max_seq_len < 131072 requests = manager.add_dummy_requests( [0], token_nums=[manager.max_num_tokens // 2], is_gen=False @@ -1011,11 +1027,11 @@ def test_full_attention_warmup_respects_allocated_budget( manager.free_resources(request) -def test_full_attention_budget_supports_cuda_graph_warmup( - _budget_limited_full_attention_manager: KVCacheManagerV2, +def test_attention_budget_supports_cuda_graph_warmup( + _budget_limited_attention_manager: KVCacheManagerV2, _budget_warmup_spec_config, ) -> None: - manager = _budget_limited_full_attention_manager + manager = _budget_limited_attention_manager # Exercise the real graph request builder: it allocates the short requests, # queries the remaining capacity, then grows the longest generation request. engine = SimpleNamespace() @@ -1306,7 +1322,7 @@ def test_extra_tokens_are_in_context_capacity() -> None: ) assert config.typical_step == BatchDesc([KVCacheDesc(capacity=258, history_length=0)]) - assert config.constraints[1] == BatchDesc([KVCacheDesc(capacity=258, history_length=0)]) + assert config.constraints[2] == BatchDesc([KVCacheDesc(capacity=258, history_length=0)]) def test_try_commit_blocks_commits_partial_block_at_context_end() -> None: From 8662968ac5ce722665eac7777135bfee7f6f60c3 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:41:59 -0700 Subject: [PATCH 06/15] [None][fix] Keep V2 budget fix focused on existing CI failures Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../kv_cache/kv_cache_manager_v2.py | 69 ---------- .../_torch/pyexecutor/model_engine.py | 128 +++++++++--------- .../kv_cache/test_kvcm2_integration.py | 93 ++++--------- 3 files changed, 94 insertions(+), 196 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py index 84ae6d067c46..bdb642a35d39 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py @@ -3234,75 +3234,6 @@ def get_num_available_tokens( clamped = min(clamped, self._gpu_max_tokens - extra_tokens) return clamped - def get_warmup_token_capacity( - self, - *, - token_num_upper_bound: int, - max_num_draft_tokens: int = 0, - draft_kv_cache_manager: Optional["KVCacheManagerV2"] = None, - ) -> int: - """Return an input length that fits one additional generation dummy. - - Other dummies must already be resident. This query is for exclusive - warmup, not concurrent request admission, and never grows the pool. - The descriptors match both resize calls in ``add_dummy_requests``: - the target keeps generation history, while a coupled draft cache - materializes the entire input. All sequence/position limits must be - applied to the upper bound before calling this method. - """ - managers = [self] - if draft_kv_cache_manager is not None: - managers.append(draft_kv_cache_manager) - available_slots = [] - overhead = self.num_extra_kv_tokens + max_num_draft_tokens + 1 - upper = token_num_upper_bound - for manager in managers: - statistics = manager._get_storage_statistics(GPU_LEVEL) - max_util = manager.kv_cache_manager_py_config.max_util_for_resume - if any(stat.total and stat.unavailable / stat.total > max_util for stat in statistics): - return 0 - available_slots.append([stat.available for stat in statistics]) - if manager._gpu_max_tokens is not None: - upper = min(upper, manager._gpu_max_tokens - overhead) - minimum_tokens = 2 if self._has_cp_helix else 1 - if upper < minimum_tokens: - return 0 - - def fits(tokens: int) -> bool: - for index, (manager, available) in enumerate(zip(managers, available_slots)): - materialize_history = index != 0 - descriptor = KVCacheDesc( - capacity=tokens + overhead, - history_length=0 if materialize_history else tokens - 1, - ) - needed = _introspection.compute_slots_for_batch( - manager.impl, - BatchDesc([descriptor]), - manager._ledger_tokens_per_block, - manager.kv_cache_manager_py_config.swa_scratch_reuse - if materialize_history - else None, - ) - if any(required > free for required, free in zip(needed, available)): - return False - return True - - if fits(upper): - return upper - if not fits(minimum_tokens): - return 0 - # SWA retention can oscillate by one slot within a page. Search one - # common page phase, checking target and draft at the same input length. - page = math.lcm(*(manager._ledger_tokens_per_block for manager in managers)) - lo, hi = 0, upper // page + 1 - while hi - lo > 1: - mid = (lo + hi) // 2 - if fits(mid * page): - lo = mid - else: - hi = mid - return max(minimum_tokens, lo * page) - def get_num_free_blocks(self) -> int: # NOTE This method is used to get the number of blocks in the primary pool not the FREE blocks. # However, since we only use this function when the kv cache manager is empty, so it is safe to do so. diff --git a/tensorrt_llm/_torch/pyexecutor/model_engine.py b/tensorrt_llm/_torch/pyexecutor/model_engine.py index 06c348b44ebd..64c0290996e1 100644 --- a/tensorrt_llm/_torch/pyexecutor/model_engine.py +++ b/tensorrt_llm/_torch/pyexecutor/model_engine.py @@ -3401,6 +3401,73 @@ def _create_cuda_graph_warmup_request( if num_mixed_contexts >= batch_size: return None + # Add one dummy request with the maximum possible sequence length. + max_seq_len = min( + self.max_seq_len if max_seq_len is None else max_seq_len, + kv_cache_manager.max_seq_len) + + # Use max_draft_loop_tokens for capacity estimation to account + # for the actual KV reservation per request. + _kv_draft = self.max_draft_loop_tokens + + # Determine the input bound before allocating the warmup batch. + token_num = max_seq_len - 1 - get_num_extra_kv_tokens( + self.spec_config) - _kv_draft + model_config = self.model.model_config.pretrained_config + max_position_embeddings = getattr(model_config, + 'max_position_embeddings', None) + if is_enc_dec: + # For enc-dec models the engine max_seq_len covers the encoder + # sequence, which may exceed the decoder's position table (e.g. + # Whisper: 1500 encoder positions vs max_target_positions=448). + decoder_position_limit = getattr(model_config, + 'max_target_positions', None) + if decoder_position_limit is not None: + max_position_embeddings = ( + decoder_position_limit if max_position_embeddings is None + else min(max_position_embeddings, decoder_position_limit)) + if max_position_embeddings is not None: + token_num = min(token_num, max_position_embeddings - _kv_draft) + + if isinstance(kv_cache_manager, KVCacheManagerV2): + # V2 adds one generation token beyond the draft/extra reservation. + # Include that token in the existing query, then return input length. + available_tokens = kv_cache_manager.get_num_available_tokens( + token_num_upper_bound=token_num + 1, + batch_size=batch_size, + max_num_draft_tokens=_kv_draft) + if draft_kv_cache_manager is not None: + available_tokens = min( + available_tokens, + draft_kv_cache_manager.get_num_available_tokens( + token_num_upper_bound=token_num + 1, + batch_size=batch_size, + max_num_draft_tokens=_kv_draft)) + token_num = min(token_num, available_tokens - 1) + minimum_tokens = ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1 + if token_num < minimum_tokens: + return None + else: + available_tokens = kv_cache_manager.get_num_available_tokens( + token_num_upper_bound=max_seq_len, + batch_size=batch_size, + max_num_draft_tokens=_kv_draft) + if draft_kv_cache_manager is not None: + available_tokens = min( + available_tokens, + draft_kv_cache_manager.get_num_available_tokens( + token_num_upper_bound=max_seq_len, + batch_size=batch_size, + max_num_draft_tokens=_kv_draft)) + token_num = max( + ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1, + min(token_num, available_tokens)) + if max_position_embeddings is not None: + token_num = min(token_num, max_position_embeddings - _kv_draft) + + token_num = int( + token_num) # Ensure int for range() in add_dummy_requests + # Add (batch_size - 1) dummy requests with the minimal sequence # length. Mixed capture must create its context rows as real context # requests; converting generation dummies afterward leaves their @@ -3470,67 +3537,6 @@ def free_warmup_requests() -> None: if draft_kv_cache_manager is not None: draft_kv_cache_manager.free_resources(r) - # Add one dummy request with the maximum possible sequence length. - max_seq_len = min( - self.max_seq_len if max_seq_len is None else max_seq_len, - kv_cache_manager.max_seq_len) - - # Use max_draft_loop_tokens for capacity estimation to account - # for the actual KV reservation per request. - _kv_draft = self.max_draft_loop_tokens - - # Apply every static limit before the V2 query. Changing its result - # afterward can increase SWA page demand at a different page phase. - token_num = max_seq_len - 1 - get_num_extra_kv_tokens( - self.spec_config) - _kv_draft - model_config = self.model.model_config.pretrained_config - max_position_embeddings = getattr(model_config, - 'max_position_embeddings', None) - if is_enc_dec: - # For enc-dec models the engine max_seq_len covers the encoder - # sequence, which may exceed the decoder's position table (e.g. - # Whisper: 1500 encoder positions vs max_target_positions=448). - decoder_position_limit = getattr(model_config, - 'max_target_positions', None) - if decoder_position_limit is not None: - max_position_embeddings = ( - decoder_position_limit if max_position_embeddings is None - else min(max_position_embeddings, decoder_position_limit)) - if max_position_embeddings is not None: - token_num = min(token_num, max_position_embeddings - _kv_draft) - - if isinstance(kv_cache_manager, KVCacheManagerV2): - assert draft_kv_cache_manager is None or isinstance( - draft_kv_cache_manager, KVCacheManagerV2) - token_num = kv_cache_manager.get_warmup_token_capacity( - token_num_upper_bound=token_num, - max_num_draft_tokens=_kv_draft, - draft_kv_cache_manager=draft_kv_cache_manager) - minimum_tokens = ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1 - if token_num < minimum_tokens: - free_warmup_requests() - return None - else: - available_tokens = kv_cache_manager.get_num_available_tokens( - token_num_upper_bound=max_seq_len, - batch_size=batch_size, - max_num_draft_tokens=_kv_draft) - if draft_kv_cache_manager is not None: - available_tokens = min( - available_tokens, - draft_kv_cache_manager.get_num_available_tokens( - token_num_upper_bound=max_seq_len, - batch_size=batch_size, - max_num_draft_tokens=_kv_draft)) - token_num = max( - ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1, - min(token_num, available_tokens)) - if max_position_embeddings is not None: - token_num = min(token_num, max_position_embeddings - _kv_draft) - - token_num = int( - token_num) # Ensure int for range() in add_dummy_requests - max_seq_len_request = kv_cache_manager.add_dummy_requests( request_ids=[batch_size - 1], token_nums=[token_num], diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index b2f0d0baed28..99a712b63425 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -15,6 +15,7 @@ import array import os +from collections.abc import Iterator from dataclasses import dataclass, field, replace from types import SimpleNamespace from unittest.mock import Mock, call, patch @@ -892,18 +893,11 @@ def test_default_uses_allocator_fallback() -> None: @pytest.mark.parametrize( - "kv_cache_type,attention_windows,generation_capacity", - [ - (CacheType.SELF, [None, None], 1024), - (CacheType.SELF, [None, 256], 1024), - (CacheType.CROSS, [None, None], 1024), - (CacheType.SELFKONLY, [None, None], 1024), - ], - ids=["full_attention", "mixed_attention", "cross_attention", "key_only"], + "kv_cache_type", + [CacheType.SELF, CacheType.SELFKONLY], + ids=["full_attention", "key_only"], ) -def test_avg_seq_len_builds_warmup_constraints( - kv_cache_type, attention_windows, generation_capacity -) -> None: +def test_avg_seq_len_builds_warmup_constraints(kv_cache_type: CacheType) -> None: config = _make_cache_config_for_test( KvCacheConfig(use_kv_cache_manager_v2=True, host_cache_size=0, avg_seq_len=1024), kv_cache_type=kv_cache_type, @@ -911,7 +905,6 @@ def test_avg_seq_len_builds_warmup_constraints( max_seq_len=1024, max_num_tokens=2048, max_draft_len=2, - max_attention_window_vec=attention_windows, ) assert config.typical_step == BatchDesc( @@ -921,7 +914,7 @@ def test_avg_seq_len_builds_warmup_constraints( assert config.constraints == [ BatchDesc( [ - KVCacheDesc(capacity=generation_capacity, history_length=generation_capacity - 1), + KVCacheDesc(capacity=1024, history_length=1023), ] ), BatchDesc([KVCacheDesc(capacity=3, history_length=0)] * 3), @@ -929,52 +922,25 @@ def test_avg_seq_len_builds_warmup_constraints( ] -@pytest.fixture(params=[0, 16], ids=["decode", "speculative"]) -def _budget_warmup_spec_config(request: pytest.FixtureRequest): - return ( - Eagle3DecodingConfig(max_draft_len=request.param, speculative_model="dummy") - if request.param - else None - ) - - -@pytest.fixture(params=[False, True], ids=["no_guard", "guard"]) -def _budget_warmup_guard_page(request: pytest.FixtureRequest, monkeypatch): - if request.param: - monkeypatch.setenv("TRTLLM_KV_GUARD_PAGE", "1") - else: - monkeypatch.delenv("TRTLLM_KV_GUARD_PAGE", raising=False) - return request.param - - -@pytest.fixture(params=[False, True], ids=["final", "estimation"]) -def _budget_warmup_phase(request: pytest.FixtureRequest): - return request.param - - -@pytest.fixture(params=[None, [256, 256], [256, 131072]], ids=["full", "swa", "mixed"]) -def _budget_warmup_windows(request: pytest.FixtureRequest): - return request.param - - -@pytest.fixture(params=["bytes", "tokens"]) -def _budget_limited_attention_manager( +@pytest.fixture( + params=[("bytes", False), ("bytes", True), ("tokens", False), ("tokens", True)], + ids=["bytes-final", "bytes-estimation", "tokens-final", "tokens-estimation"], +) +def _budget_limited_full_attention_manager( request: pytest.FixtureRequest, - _budget_warmup_spec_config, - _budget_warmup_guard_page, - _budget_warmup_phase, - _budget_warmup_windows, -): + monkeypatch: pytest.MonkeyPatch, +) -> Iterator[KVCacheManagerV2]: if not torch.cuda.is_available(): pytest.skip("requires CUDA") init_cuda_once() - budget = {"max_gpu_total_bytes": 17 << 20} if request.param == "bytes" else {"max_tokens": 8192} + monkeypatch.delenv("TRTLLM_KV_GUARD_PAGE", raising=False) + budget_type, is_estimating = request.param + budget = {"max_gpu_total_bytes": 17 << 20} if budget_type == "bytes" else {"max_tokens": 8192} config = KvCacheConfig( use_kv_cache_manager_v2=True, enable_block_reuse=False, host_cache_size=0, avg_seq_len=32768, - max_attention_window=_budget_warmup_windows, **budget, ) manager = KVCacheManagerV2( @@ -989,28 +955,25 @@ def _budget_limited_attention_manager( max_num_tokens=2048, mapping=Mapping(), dtype=DataType.HALF, - spec_config=_budget_warmup_spec_config, - is_estimating_kv_cache=_budget_warmup_phase, + is_estimating_kv_cache=is_estimating, ) try: - assert len(manager.kv_cache_map) == int(_budget_warmup_guard_page) + assert not manager.kv_cache_map yield manager finally: manager.shutdown() -def test_attention_warmup_respects_allocated_budget( - _budget_limited_attention_manager: KVCacheManagerV2, +def test_full_attention_warmup_respects_allocated_budget( + _budget_limited_full_attention_manager: KVCacheManagerV2, ) -> None: - manager = _budget_limited_attention_manager + manager = _budget_limited_full_attention_manager requested_quota = manager.kv_cache_manager_py_config.cache_tiers[0].quota allocated_bytes = manager.impl.get_quota(kv_cache_v2_module.GPU_LEVEL) # These small quotas round up to a 2 MiB GPU allocation grain. The model's # full context requires at least 256 MiB, far beyond either configured budget. assert 0 < allocated_bytes <= requested_quota + (2 << 20) - assert manager.max_num_tokens < manager.max_seq_len <= 131072 - if any(window is None for window in manager.max_attention_window_vec): - assert manager.max_seq_len < 131072 + assert manager.max_num_tokens < manager.max_seq_len < 131072 requests = manager.add_dummy_requests( [0], token_nums=[manager.max_num_tokens // 2], is_gen=False @@ -1027,16 +990,14 @@ def test_attention_warmup_respects_allocated_budget( manager.free_resources(request) -def test_attention_budget_supports_cuda_graph_warmup( - _budget_limited_attention_manager: KVCacheManagerV2, - _budget_warmup_spec_config, +def test_full_attention_budget_supports_cuda_graph_warmup( + _budget_limited_full_attention_manager: KVCacheManagerV2, ) -> None: - manager = _budget_limited_attention_manager - # Exercise the real graph request builder: it allocates the short requests, - # queries the remaining capacity, then grows the longest generation request. + manager = _budget_limited_full_attention_manager + # Exercise the real graph request builder against the allocated pool budget. engine = SimpleNamespace() engine.kv_cache_manager_key = ResourceManagerType.KV_CACHE_MANAGER - engine.spec_config = _budget_warmup_spec_config + engine.spec_config = None engine.max_beam_width = 1 engine.max_draft_loop_tokens = manager.max_draft_len engine.max_seq_len = 131072 From edaa0fea23bb3fbd7f658afd5c0c387bab287184 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:59:08 -0700 Subject: [PATCH 07/15] [None][fix] Restore warmup capacity query after short request allocation Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../_torch/pyexecutor/model_engine.py | 120 ++++++++---------- 1 file changed, 53 insertions(+), 67 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/model_engine.py b/tensorrt_llm/_torch/pyexecutor/model_engine.py index 64c0290996e1..7a73d016c8d4 100644 --- a/tensorrt_llm/_torch/pyexecutor/model_engine.py +++ b/tensorrt_llm/_torch/pyexecutor/model_engine.py @@ -3401,73 +3401,6 @@ def _create_cuda_graph_warmup_request( if num_mixed_contexts >= batch_size: return None - # Add one dummy request with the maximum possible sequence length. - max_seq_len = min( - self.max_seq_len if max_seq_len is None else max_seq_len, - kv_cache_manager.max_seq_len) - - # Use max_draft_loop_tokens for capacity estimation to account - # for the actual KV reservation per request. - _kv_draft = self.max_draft_loop_tokens - - # Determine the input bound before allocating the warmup batch. - token_num = max_seq_len - 1 - get_num_extra_kv_tokens( - self.spec_config) - _kv_draft - model_config = self.model.model_config.pretrained_config - max_position_embeddings = getattr(model_config, - 'max_position_embeddings', None) - if is_enc_dec: - # For enc-dec models the engine max_seq_len covers the encoder - # sequence, which may exceed the decoder's position table (e.g. - # Whisper: 1500 encoder positions vs max_target_positions=448). - decoder_position_limit = getattr(model_config, - 'max_target_positions', None) - if decoder_position_limit is not None: - max_position_embeddings = ( - decoder_position_limit if max_position_embeddings is None - else min(max_position_embeddings, decoder_position_limit)) - if max_position_embeddings is not None: - token_num = min(token_num, max_position_embeddings - _kv_draft) - - if isinstance(kv_cache_manager, KVCacheManagerV2): - # V2 adds one generation token beyond the draft/extra reservation. - # Include that token in the existing query, then return input length. - available_tokens = kv_cache_manager.get_num_available_tokens( - token_num_upper_bound=token_num + 1, - batch_size=batch_size, - max_num_draft_tokens=_kv_draft) - if draft_kv_cache_manager is not None: - available_tokens = min( - available_tokens, - draft_kv_cache_manager.get_num_available_tokens( - token_num_upper_bound=token_num + 1, - batch_size=batch_size, - max_num_draft_tokens=_kv_draft)) - token_num = min(token_num, available_tokens - 1) - minimum_tokens = ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1 - if token_num < minimum_tokens: - return None - else: - available_tokens = kv_cache_manager.get_num_available_tokens( - token_num_upper_bound=max_seq_len, - batch_size=batch_size, - max_num_draft_tokens=_kv_draft) - if draft_kv_cache_manager is not None: - available_tokens = min( - available_tokens, - draft_kv_cache_manager.get_num_available_tokens( - token_num_upper_bound=max_seq_len, - batch_size=batch_size, - max_num_draft_tokens=_kv_draft)) - token_num = max( - ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1, - min(token_num, available_tokens)) - if max_position_embeddings is not None: - token_num = min(token_num, max_position_embeddings - _kv_draft) - - token_num = int( - token_num) # Ensure int for range() in add_dummy_requests - # Add (batch_size - 1) dummy requests with the minimal sequence # length. Mixed capture must create its context rows as real context # requests; converting generation dummies afterward leaves their @@ -3537,6 +3470,59 @@ def free_warmup_requests() -> None: if draft_kv_cache_manager is not None: draft_kv_cache_manager.free_resources(r) + # Add one dummy request with the maximum possible sequence length. + max_seq_len = min( + self.max_seq_len if max_seq_len is None else max_seq_len, + kv_cache_manager.max_seq_len) + + # Use max_draft_loop_tokens for capacity estimation to account + # for the actual KV reservation per request. + _kv_draft = self.max_draft_loop_tokens + available_tokens = kv_cache_manager.get_num_available_tokens( + token_num_upper_bound=max_seq_len, + batch_size=batch_size, + max_num_draft_tokens=_kv_draft) + + # Also consider draft KV cache capacity when it exists + if draft_kv_cache_manager is not None: + draft_available_tokens = draft_kv_cache_manager.get_num_available_tokens( + batch_size=batch_size, + token_num_upper_bound=max_seq_len, + max_num_draft_tokens=_kv_draft) + available_tokens = min(available_tokens, draft_available_tokens) + + if isinstance(kv_cache_manager, KVCacheManagerV2): + # V2 reserves one generation token beyond the draft/extra tokens. + available_tokens -= 1 + minimum_tokens = ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1 + if available_tokens < minimum_tokens: + free_warmup_requests() + return None + + token_num = max( + ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1, + min( + available_tokens, max_seq_len - 1 - + get_num_extra_kv_tokens(self.spec_config) - _kv_draft)) + model_config = self.model.model_config.pretrained_config + max_position_embeddings = getattr(model_config, + 'max_position_embeddings', None) + if is_enc_dec: + # For enc-dec models the engine max_seq_len covers the encoder + # sequence, which may exceed the decoder's position table (e.g. + # Whisper: 1500 encoder positions vs max_target_positions=448). + decoder_position_limit = getattr(model_config, + 'max_target_positions', None) + if decoder_position_limit is not None: + max_position_embeddings = ( + decoder_position_limit if max_position_embeddings is None + else min(max_position_embeddings, decoder_position_limit)) + if max_position_embeddings is not None: + token_num = min(token_num, max_position_embeddings - _kv_draft) + + token_num = int( + token_num) # Ensure int for range() in add_dummy_requests + max_seq_len_request = kv_cache_manager.add_dummy_requests( request_ids=[batch_size - 1], token_nums=[token_num], From e3eaecf00707fb822803663223ebe901e34261be Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Wed, 16 Sep 2026 18:18:46 -0700 Subject: [PATCH 08/15] [None][test] Cover insufficient CUDA graph warmup capacity Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../kv_cache/test_kvcm2_integration.py | 31 +++++++++++++++++-- 1 file changed, 28 insertions(+), 3 deletions(-) diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index 99a712b63425..3be3f7189534 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -990,8 +990,10 @@ def test_full_attention_warmup_respects_allocated_budget( manager.free_resources(request) +@pytest.mark.parametrize("max_seq_len", [None, 1], ids=["sufficient", "insufficient"]) def test_full_attention_budget_supports_cuda_graph_warmup( _budget_limited_full_attention_manager: KVCacheManagerV2, + max_seq_len: int | None, ) -> None: manager = _budget_limited_full_attention_manager # Exercise the real graph request builder against the allocated pool budget. @@ -1010,9 +1012,32 @@ def test_full_attention_budget_supports_cuda_graph_warmup( ) resources = ResourceManager({ResourceManagerType.KV_CACHE_MANAGER: manager}) - batch = PyTorchModelEngine._create_cuda_graph_warmup_request( - engine, resources, batch_size=manager.max_batch_size, draft_len=manager.max_draft_len - ) + with ( + patch.object( + manager, "get_num_available_tokens", wraps=manager.get_num_available_tokens + ) as get_available_tokens, + patch.object(manager, "free_resources", wraps=manager.free_resources) as free_resources, + ): + batch = PyTorchModelEngine._create_cuda_graph_warmup_request( + engine, + resources, + batch_size=manager.max_batch_size, + draft_len=manager.max_draft_len, + max_seq_len=max_seq_len, + ) + if max_seq_len == 1: + # The one-token budget cannot hold both the prompt and V2's extra + # generation token. Short requests must be allocated, then freed. + get_available_tokens.assert_called_once_with( + token_num_upper_bound=1, + batch_size=manager.max_batch_size, + max_num_draft_tokens=manager.max_draft_len, + ) + assert batch is None + assert free_resources.call_count == manager.max_batch_size - 1 + assert not manager.kv_cache_map + return + assert batch is not None requests = list(batch.generation_requests) try: From e4820bd2e0b1d3e7c004ae8cbd06ae964176fd72 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:56:09 +0000 Subject: [PATCH 09/15] [None][fix] Keep maximum-length warmup constraints specific to DeepSeek V4 Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../sparse/deepseek_v4/cache_manager.py | 34 ++++++++++- .../kv_cache/kv_cache_manager_v2.py | 46 --------------- .../kv_cache/mamba_cache_manager.py | 5 +- .../test_deepseek_v4_cache_manager.py | 57 +++++++++++++++++-- .../kv_cache/test_kvcm2_integration.py | 22 ++++--- 5 files changed, 96 insertions(+), 68 deletions(-) diff --git a/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py b/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py index a007443a7403..6708d8ffbf5a 100644 --- a/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py +++ b/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py @@ -47,10 +47,12 @@ from tensorrt_llm.runtime.kv_cache_manager_v2 import ( BAD_PAGE_INDEX, AttentionLayerConfig, + BatchDesc, BufferConfig, BufferId, CudaStream, DataRole, + KVCacheDesc, LayerId, PageIndexMode, ScratchDesc, @@ -1062,7 +1064,7 @@ def _get_max_tokens_from_quota(self, quota: int) -> float: def _build_cache_config(self, config: KVCacheManagerConfigPy) -> KVCacheManagerConfigPy: """ - Add DeepSeek-V4 layers to the cache config. + Add DeepSeek-V4 layers and warmup constraints to the cache config. """ layers: List[AttentionLayerConfig] = [] layer_attn_to_layer_id: Dict[Tuple[int, DeepseekV4AttentionType], LayerId] = {} @@ -1213,9 +1215,39 @@ def _add_layer( # number of layers in the KVCacheManagerPy self._num_manager_layers = len(layers) + constraints = [] + if config.initial_pool_ratio is None: + # DeepSeek-V4's windowed and compressed pools must support both + # the longest decode request and the context warmup workload. + min_decode_capacity = 1 + self.max_draft_len + self.num_extra_kv_tokens + constraints.append( + BatchDesc( + [ + KVCacheDesc( + capacity=self.max_seq_len, + history_length=self.max_seq_len - 1, + ) + ] + + [KVCacheDesc(capacity=min_decode_capacity, history_length=0)] + * (self.max_batch_size - 1) + ) + ) + if self.max_num_tokens is not None: + constraints.append( + BatchDesc( + [ + KVCacheDesc( + capacity=self.max_num_tokens + self.num_extra_kv_tokens, + history_length=0, + ) + ] + ) + ) + return replace( config, layers=layers, + constraints=constraints, ) def _init_indexer_dtype(self, sparse_attn_config: DeepSeekV4SparseAttentionConfig) -> None: diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py index bdb642a35d39..14c3c235c265 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py @@ -2769,7 +2769,6 @@ def _build_base_config( scratch_reuse_config = SwaScratchReuseConfig(max_rewind_len=self.num_extra_kv_tokens) typical_step = None - constraints = [] if kv_cache_config.pool_ratio is None: typical_seq_len = self._get_typical_seq_len(kv_cache_config) if typical_seq_len is not None and typical_seq_len > self.max_seq_len: @@ -2798,50 +2797,6 @@ def _build_base_config( * (generation_request_capacity - 1) ) - min_decode_capacity = 1 + self.max_draft_len + self.num_extra_kv_tokens - gpu_quota = next( - tier.quota for tier in cache_tiers if isinstance(tier, GpuCacheTierConfig) - ) - # Native minimum slot counts are divided by the resume watermark. - # Normalize the quota before estimating a feasible long request. - estimate = self._get_max_tokens_from_quota( - int(gpu_quota * kv_cache_config.max_util_for_resume) - ) - generation_capacity = int(min(self.max_seq_len, max(min_decode_capacity, estimate))) - # These are independent workloads. Graph warmup shortens its long - # request after allocating the short requests; requiring both at - # this estimated length would count their memory twice. - constraints.extend( - [ - BatchDesc( - [ - KVCacheDesc( - capacity=generation_capacity, - history_length=generation_capacity - 1, - ) - ] - ), - BatchDesc( - [KVCacheDesc(capacity=min_decode_capacity, history_length=0)] - * self.max_batch_size - ), - ] - ) - - # General and chunked-prefill warmup uses one fresh context request - # at the per-iteration token budget. - if self.max_num_tokens is not None: - constraints.append( - BatchDesc( - [ - KVCacheDesc( - capacity=self.max_num_tokens + self.num_extra_kv_tokens, - history_length=0, - ) - ] - ) - ) - # Subclasses (e.g. MiniMax-M3 sparse cache) can register additional # per-layer BufferConfig entries — for example a sparse index-K # buffer — without overriding the K/V/NVFP4 scale wiring above. @@ -2886,7 +2841,6 @@ def _build_base_config( cache_tiers=cache_tiers, layers=layer_configs, typical_step=typical_step, - constraints=constraints, max_util_for_resume=kv_cache_config.max_util_for_resume, enable_partial_reuse=kv_cache_config.enable_partial_reuse, # Keep the lookahead evidence and its backoff in the same tree match. diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/mamba_cache_manager.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/mamba_cache_manager.py index ad417618950b..2e40eb01508f 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/mamba_cache_manager.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/mamba_cache_manager.py @@ -3998,9 +3998,8 @@ def _build_cache_config( # The recurrent (SSM) state pool must hold one slot per resident # sequence plus every reserved dummy slot. Unlike attention pages, a # Mamba state is fixed-size per sequence, so this floor is independent - # of sequence length. The base config only emits constraints when - # ``avg_seq_len`` is set, and speculative decoding inflates the reserved - # dummy slots (CUDA-graph padding), so without an explicit floor the SSM + # of sequence length. Speculative decoding inflates the reserved dummy + # slots (CUDA-graph padding), so without this explicit floor the SSM # pool can be undersized (see the live/dummy-slot check in _setup_states # / __init__). Add a min-slots constraint of zero-capacity requests: # these cost no attention pages but reserve one SSM slot each. diff --git a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py index 3d757b10ec38..e3c4d02c23c2 100644 --- a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py +++ b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py @@ -709,6 +709,7 @@ def _create_deepseek_v4_cache_manager( enable_swa_scratch_reuse: bool = True, cold_page_codec_provider: object | None = None, host_cache_size: int | None = None, + avg_seq_len: int | None = None, ) -> Tuple[DeepseekV4CacheManager, DeepSeekV4SparseAttentionConfig]: """Helper to create a DeepseekV4CacheManager for testing.""" @@ -732,6 +733,7 @@ def _create_deepseek_v4_cache_manager( event_buffer_max_size=0, enable_swa_scratch_reuse=enable_swa_scratch_reuse, host_cache_size=host_cache_size, + avg_seq_len=avg_seq_len, ) # Create mapping (single GPU, no parallelism) @@ -1580,7 +1582,10 @@ def _assert_cache_equal( msg=f"Mismatch for layer {layer_idx}, attention type {attn_type.name} (scales)", ) - def test_max_num_tokens_is_used_by_base_config(self): + @pytest.mark.parametrize("avg_seq_len", [None, 256], ids=["default", "explicit_avg"]) + def test_warmup_constraints_preserve_deepseek_v4_capacity( + self, scratch_reuse_enabled: bool, avg_seq_len: int | None + ) -> None: max_batch_size = 2 max_seq_len = 1024 max_input_len = 127 @@ -1593,14 +1598,54 @@ def test_max_num_tokens_is_used_by_base_config(self): compress_ratios=[1, 4], dtype=DataType.BF16, compressor_dtype=DataType.FLOAT, + enable_swa_scratch_reuse=scratch_reuse_enabled, + avg_seq_len=avg_seq_len, ) - assert cache_manager.kv_cache_manager_py_config.typical_step == BatchDesc( - [ - KVCacheDesc(capacity=max_num_tokens, history_length=0), - KVCacheDesc(capacity=max_seq_len, history_length=max_seq_len - 1), + requests = [] + try: + config = cache_manager.kv_cache_manager_py_config + typical_seq_len = max_seq_len if avg_seq_len is None else avg_seq_len + assert config.typical_step == BatchDesc( + [ + KVCacheDesc(capacity=max_num_tokens, history_length=0), + KVCacheDesc(capacity=typical_seq_len, history_length=typical_seq_len - 1), + ] + ) + assert config.constraints == [ + BatchDesc( + [ + KVCacheDesc(capacity=max_seq_len, history_length=max_seq_len - 1), + KVCacheDesc(capacity=1, history_length=0), + ] + ), + BatchDesc([KVCacheDesc(capacity=max_num_tokens, history_length=0)]), ] - ) + + # Both V4 workloads must fit the native pools after the constraints + # move to the specialized manager. + generation_requests = cache_manager.add_dummy_requests( + request_ids=[0, 1], token_nums=[1, max_seq_len - 1], is_gen=True + ) + assert generation_requests is not None + requests.extend(generation_requests) + long_cache = cache_manager.kv_cache_map[requests[1].py_request_id] + assert long_cache.capacity == max_seq_len + for request in requests: + cache_manager.free_resources(request) + requests.clear() + + context_requests = cache_manager.add_dummy_requests( + request_ids=[0], token_nums=[max_num_tokens], is_gen=False + ) + assert context_requests is not None + requests.extend(context_requests) + context_cache = cache_manager.kv_cache_map[requests[0].py_request_id] + assert context_cache.capacity == max_num_tokens + finally: + for request in requests: + cache_manager.free_resources(request) + cache_manager.shutdown() def test_indexer_cache_layout_default(self): """DeepSeek-V4 defaults to FP4 indexer K cache on Blackwell+.""" diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index 3be3f7189534..3670c4eae8f2 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -219,7 +219,6 @@ def _make_cache_config_for_test( cache_manager.enable_joint_kv_cache_reuse = False cache_manager.reuse_match_backoff = 0 cache_manager.get_layer_bytes_per_token = lambda **_: 128 - cache_manager._get_max_tokens_from_quota = lambda _: max_seq_len # Mirrors __init__: without helix the ledger block equals the physical # page (the helper re-enacts construction for partial instances). cache_manager._ledger_tokens_per_block = 128 @@ -897,10 +896,16 @@ def test_default_uses_allocator_fallback() -> None: [CacheType.SELF, CacheType.SELFKONLY], ids=["full_attention", "key_only"], ) -def test_avg_seq_len_builds_warmup_constraints(kv_cache_type: CacheType) -> None: +@pytest.mark.parametrize( + "max_attention_window_vec", [[None], [128, None]], ids=["uniform", "mixed"] +) +def test_avg_seq_len_does_not_require_max_length_warmup( + kv_cache_type: CacheType, max_attention_window_vec: list[int | None] +) -> None: config = _make_cache_config_for_test( KvCacheConfig(use_kv_cache_manager_v2=True, host_cache_size=0, avg_seq_len=1024), kv_cache_type=kv_cache_type, + max_attention_window_vec=max_attention_window_vec, max_batch_size=3, max_seq_len=1024, max_num_tokens=2048, @@ -911,15 +916,7 @@ def test_avg_seq_len_builds_warmup_constraints(kv_cache_type: CacheType) -> None [KVCacheDesc(capacity=2048, history_length=0)] + [KVCacheDesc(capacity=1024, history_length=1021)] * 2 ) - assert config.constraints == [ - BatchDesc( - [ - KVCacheDesc(capacity=1024, history_length=1023), - ] - ), - BatchDesc([KVCacheDesc(capacity=3, history_length=0)] * 3), - BatchDesc([KVCacheDesc(capacity=2048, history_length=0)]), - ] + assert config.constraints == [] @pytest.fixture( @@ -973,6 +970,7 @@ def test_full_attention_warmup_respects_allocated_budget( # These small quotas round up to a 2 MiB GPU allocation grain. The model's # full context requires at least 256 MiB, far beyond either configured budget. assert 0 < allocated_bytes <= requested_quota + (2 << 20) + assert manager.kv_cache_manager_py_config.constraints == [] assert manager.max_num_tokens < manager.max_seq_len < 131072 requests = manager.add_dummy_requests( @@ -1308,7 +1306,7 @@ def test_extra_tokens_are_in_context_capacity() -> None: ) assert config.typical_step == BatchDesc([KVCacheDesc(capacity=258, history_length=0)]) - assert config.constraints[2] == BatchDesc([KVCacheDesc(capacity=258, history_length=0)]) + assert config.constraints == [] def test_try_commit_blocks_commits_partial_block_at_context_end() -> None: From 42f0f0011432b4ca045e986342cbb601be88b6af Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Sun, 20 Sep 2026 16:36:35 +0000 Subject: [PATCH 10/15] [None][test] Keep KVCM constraint test changes minimal Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../test_deepseek_v4_cache_manager.py | 57 +------ .../kv_cache/test_kvcm2_integration.py | 155 +----------------- 2 files changed, 8 insertions(+), 204 deletions(-) diff --git a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py index e3c4d02c23c2..3d757b10ec38 100644 --- a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py +++ b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py @@ -709,7 +709,6 @@ def _create_deepseek_v4_cache_manager( enable_swa_scratch_reuse: bool = True, cold_page_codec_provider: object | None = None, host_cache_size: int | None = None, - avg_seq_len: int | None = None, ) -> Tuple[DeepseekV4CacheManager, DeepSeekV4SparseAttentionConfig]: """Helper to create a DeepseekV4CacheManager for testing.""" @@ -733,7 +732,6 @@ def _create_deepseek_v4_cache_manager( event_buffer_max_size=0, enable_swa_scratch_reuse=enable_swa_scratch_reuse, host_cache_size=host_cache_size, - avg_seq_len=avg_seq_len, ) # Create mapping (single GPU, no parallelism) @@ -1582,10 +1580,7 @@ def _assert_cache_equal( msg=f"Mismatch for layer {layer_idx}, attention type {attn_type.name} (scales)", ) - @pytest.mark.parametrize("avg_seq_len", [None, 256], ids=["default", "explicit_avg"]) - def test_warmup_constraints_preserve_deepseek_v4_capacity( - self, scratch_reuse_enabled: bool, avg_seq_len: int | None - ) -> None: + def test_max_num_tokens_is_used_by_base_config(self): max_batch_size = 2 max_seq_len = 1024 max_input_len = 127 @@ -1598,54 +1593,14 @@ def test_warmup_constraints_preserve_deepseek_v4_capacity( compress_ratios=[1, 4], dtype=DataType.BF16, compressor_dtype=DataType.FLOAT, - enable_swa_scratch_reuse=scratch_reuse_enabled, - avg_seq_len=avg_seq_len, ) - requests = [] - try: - config = cache_manager.kv_cache_manager_py_config - typical_seq_len = max_seq_len if avg_seq_len is None else avg_seq_len - assert config.typical_step == BatchDesc( - [ - KVCacheDesc(capacity=max_num_tokens, history_length=0), - KVCacheDesc(capacity=typical_seq_len, history_length=typical_seq_len - 1), - ] - ) - assert config.constraints == [ - BatchDesc( - [ - KVCacheDesc(capacity=max_seq_len, history_length=max_seq_len - 1), - KVCacheDesc(capacity=1, history_length=0), - ] - ), - BatchDesc([KVCacheDesc(capacity=max_num_tokens, history_length=0)]), + assert cache_manager.kv_cache_manager_py_config.typical_step == BatchDesc( + [ + KVCacheDesc(capacity=max_num_tokens, history_length=0), + KVCacheDesc(capacity=max_seq_len, history_length=max_seq_len - 1), ] - - # Both V4 workloads must fit the native pools after the constraints - # move to the specialized manager. - generation_requests = cache_manager.add_dummy_requests( - request_ids=[0, 1], token_nums=[1, max_seq_len - 1], is_gen=True - ) - assert generation_requests is not None - requests.extend(generation_requests) - long_cache = cache_manager.kv_cache_map[requests[1].py_request_id] - assert long_cache.capacity == max_seq_len - for request in requests: - cache_manager.free_resources(request) - requests.clear() - - context_requests = cache_manager.add_dummy_requests( - request_ids=[0], token_nums=[max_num_tokens], is_gen=False - ) - assert context_requests is not None - requests.extend(context_requests) - context_cache = cache_manager.kv_cache_map[requests[0].py_request_id] - assert context_cache.capacity == max_num_tokens - finally: - for request in requests: - cache_manager.free_resources(request) - cache_manager.shutdown() + ) def test_indexer_cache_layout_default(self): """DeepSeek-V4 defaults to FP4 indexer K cache on Blackwell+.""" diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index 3670c4eae8f2..d4dfcb60d18c 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -15,7 +15,6 @@ import array import os -from collections.abc import Iterator from dataclasses import dataclass, field, replace from types import SimpleNamespace from unittest.mock import Mock, call, patch @@ -37,8 +36,6 @@ _update_kv_cache_draft_token_location, ) from tensorrt_llm._torch.pyexecutor.llm_request import LlmRequest, LlmRequestState -from tensorrt_llm._torch.pyexecutor.model_engine import PyTorchModelEngine -from tensorrt_llm._torch.pyexecutor.resource_manager import ResourceManager, ResourceManagerType from tensorrt_llm._torch.pyexecutor.scheduler import ScheduledRequests from tensorrt_llm.bindings import DataType, SamplingConfig from tensorrt_llm.bindings.BuildInfo import ENABLE_MULTI_DEVICE @@ -891,21 +888,9 @@ def test_default_uses_allocator_fallback() -> None: assert config.constraints == [] -@pytest.mark.parametrize( - "kv_cache_type", - [CacheType.SELF, CacheType.SELFKONLY], - ids=["full_attention", "key_only"], -) -@pytest.mark.parametrize( - "max_attention_window_vec", [[None], [128, None]], ids=["uniform", "mixed"] -) -def test_avg_seq_len_does_not_require_max_length_warmup( - kv_cache_type: CacheType, max_attention_window_vec: list[int | None] -) -> None: +def test_avg_seq_len_does_not_require_max_length_warmup() -> None: config = _make_cache_config_for_test( - KvCacheConfig(use_kv_cache_manager_v2=True, host_cache_size=0, avg_seq_len=1024), - kv_cache_type=kv_cache_type, - max_attention_window_vec=max_attention_window_vec, + KvCacheConfig(host_cache_size=0, avg_seq_len=1024), max_batch_size=3, max_seq_len=1024, max_num_tokens=2048, @@ -919,142 +904,6 @@ def test_avg_seq_len_does_not_require_max_length_warmup( assert config.constraints == [] -@pytest.fixture( - params=[("bytes", False), ("bytes", True), ("tokens", False), ("tokens", True)], - ids=["bytes-final", "bytes-estimation", "tokens-final", "tokens-estimation"], -) -def _budget_limited_full_attention_manager( - request: pytest.FixtureRequest, - monkeypatch: pytest.MonkeyPatch, -) -> Iterator[KVCacheManagerV2]: - if not torch.cuda.is_available(): - pytest.skip("requires CUDA") - init_cuda_once() - monkeypatch.delenv("TRTLLM_KV_GUARD_PAGE", raising=False) - budget_type, is_estimating = request.param - budget = {"max_gpu_total_bytes": 17 << 20} if budget_type == "bytes" else {"max_tokens": 8192} - config = KvCacheConfig( - use_kv_cache_manager_v2=True, - enable_block_reuse=False, - host_cache_size=0, - avg_seq_len=32768, - **budget, - ) - manager = KVCacheManagerV2( - config, - CacheType.SELF, - num_layers=2, - num_kv_heads=2, - head_dim=128, - tokens_per_block=32, - max_seq_len=131072, - max_batch_size=8, - max_num_tokens=2048, - mapping=Mapping(), - dtype=DataType.HALF, - is_estimating_kv_cache=is_estimating, - ) - try: - assert not manager.kv_cache_map - yield manager - finally: - manager.shutdown() - - -def test_full_attention_warmup_respects_allocated_budget( - _budget_limited_full_attention_manager: KVCacheManagerV2, -) -> None: - manager = _budget_limited_full_attention_manager - requested_quota = manager.kv_cache_manager_py_config.cache_tiers[0].quota - allocated_bytes = manager.impl.get_quota(kv_cache_v2_module.GPU_LEVEL) - # These small quotas round up to a 2 MiB GPU allocation grain. The model's - # full context requires at least 256 MiB, far beyond either configured budget. - assert 0 < allocated_bytes <= requested_quota + (2 << 20) - assert manager.kv_cache_manager_py_config.constraints == [] - assert manager.max_num_tokens < manager.max_seq_len < 131072 - - requests = manager.add_dummy_requests( - [0], token_nums=[manager.max_num_tokens // 2], is_gen=False - ) - assert requests is not None - try: - cache = manager.kv_cache_map[requests[0].py_request_id] - assert cache.resize(manager.max_num_tokens, history_length=0) - assert cache.capacity == manager.max_num_tokens - cache.suspend() - assert cache.resume(torch.cuda.current_stream().cuda_stream) - finally: - for request in requests: - manager.free_resources(request) - - -@pytest.mark.parametrize("max_seq_len", [None, 1], ids=["sufficient", "insufficient"]) -def test_full_attention_budget_supports_cuda_graph_warmup( - _budget_limited_full_attention_manager: KVCacheManagerV2, - max_seq_len: int | None, -) -> None: - manager = _budget_limited_full_attention_manager - # Exercise the real graph request builder against the allocated pool budget. - engine = SimpleNamespace() - engine.kv_cache_manager_key = ResourceManagerType.KV_CACHE_MANAGER - engine.spec_config = None - engine.max_beam_width = 1 - engine.max_draft_loop_tokens = manager.max_draft_len - engine.max_seq_len = 131072 - engine.use_mrope = False - engine.get_runtime_tokens_per_gen_step = lambda draft_len: draft_len + 1 - engine._get_draft_kv_cache_manager = lambda _: None - engine._is_encoder_decoder_model = lambda: False - engine.model = SimpleNamespace( - model_config=SimpleNamespace(pretrained_config=SimpleNamespace()) - ) - resources = ResourceManager({ResourceManagerType.KV_CACHE_MANAGER: manager}) - - with ( - patch.object( - manager, "get_num_available_tokens", wraps=manager.get_num_available_tokens - ) as get_available_tokens, - patch.object(manager, "free_resources", wraps=manager.free_resources) as free_resources, - ): - batch = PyTorchModelEngine._create_cuda_graph_warmup_request( - engine, - resources, - batch_size=manager.max_batch_size, - draft_len=manager.max_draft_len, - max_seq_len=max_seq_len, - ) - if max_seq_len == 1: - # The one-token budget cannot hold both the prompt and V2's extra - # generation token. Short requests must be allocated, then freed. - get_available_tokens.assert_called_once_with( - token_num_upper_bound=1, - batch_size=manager.max_batch_size, - max_num_draft_tokens=manager.max_draft_len, - ) - assert batch is None - assert free_resources.call_count == manager.max_batch_size - 1 - assert not manager.kv_cache_map - return - - assert batch is not None - requests = list(batch.generation_requests) - try: - assert len(requests) == manager.max_batch_size - longest_request = requests[0] - cache = manager.kv_cache_map[longest_request.py_request_id] - assert cache.capacity > manager.max_num_tokens - manager.free_resources(longest_request) - requests.remove(longest_request) - # Once the longest request finishes, another generation request can - # grow across page boundaries into the released capacity. - cache = manager.kv_cache_map[requests[0].py_request_id] - assert cache.capacity < manager.max_num_tokens - assert cache.resize(manager.max_num_tokens, history_length=cache.history_length + 1) - finally: - for request in requests: - manager.free_resources(request) - - def test_avg_seq_len_updates_typical_step() -> None: config = _make_cache_config_for_test( KvCacheConfig(avg_seq_len=256), From 3ca96e87c654d8b3e700aac8a4f40fb68b914720 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Sun, 20 Sep 2026 16:46:05 +0000 Subject: [PATCH 11/15] [None][fix] Preserve generic context warmup constraints Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../sparse/deepseek_v4/cache_manager.py | 18 +++--------------- .../pyexecutor/kv_cache/kv_cache_manager_v2.py | 16 ++++++++++++++++ .../pyexecutor/kv_cache/mamba_cache_manager.py | 5 +++-- .../kv_cache/test_kvcm2_integration.py | 6 +++--- 4 files changed, 25 insertions(+), 20 deletions(-) diff --git a/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py b/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py index 6708d8ffbf5a..b528380ff0a7 100644 --- a/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py +++ b/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py @@ -1215,10 +1215,10 @@ def _add_layer( # number of layers in the KVCacheManagerPy self._num_manager_layers = len(layers) - constraints = [] + constraints = list(config.constraints) if config.initial_pool_ratio is None: - # DeepSeek-V4's windowed and compressed pools must support both - # the longest decode request and the context warmup workload. + # DeepSeek-V4's windowed and compressed pools must also support + # the longest decode request alongside the short decode requests. min_decode_capacity = 1 + self.max_draft_len + self.num_extra_kv_tokens constraints.append( BatchDesc( @@ -1232,18 +1232,6 @@ def _add_layer( * (self.max_batch_size - 1) ) ) - if self.max_num_tokens is not None: - constraints.append( - BatchDesc( - [ - KVCacheDesc( - capacity=self.max_num_tokens + self.num_extra_kv_tokens, - history_length=0, - ) - ] - ) - ) - return replace( config, layers=layers, diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py index 14c3c235c265..2e572129eb00 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py @@ -2769,6 +2769,7 @@ def _build_base_config( scratch_reuse_config = SwaScratchReuseConfig(max_rewind_len=self.num_extra_kv_tokens) typical_step = None + constraints = [] if kv_cache_config.pool_ratio is None: typical_seq_len = self._get_typical_seq_len(kv_cache_config) if typical_seq_len is not None and typical_seq_len > self.max_seq_len: @@ -2797,6 +2798,20 @@ def _build_base_config( * (generation_request_capacity - 1) ) + # General and chunked-prefill warmup uses one fresh context request + # at the per-iteration token budget. + if self.max_num_tokens is not None: + constraints.append( + BatchDesc( + [ + KVCacheDesc( + capacity=self.max_num_tokens + self.num_extra_kv_tokens, + history_length=0, + ) + ] + ) + ) + # Subclasses (e.g. MiniMax-M3 sparse cache) can register additional # per-layer BufferConfig entries — for example a sparse index-K # buffer — without overriding the K/V/NVFP4 scale wiring above. @@ -2841,6 +2856,7 @@ def _build_base_config( cache_tiers=cache_tiers, layers=layer_configs, typical_step=typical_step, + constraints=constraints, max_util_for_resume=kv_cache_config.max_util_for_resume, enable_partial_reuse=kv_cache_config.enable_partial_reuse, # Keep the lookahead evidence and its backoff in the same tree match. diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/mamba_cache_manager.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/mamba_cache_manager.py index 2e40eb01508f..ad417618950b 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/mamba_cache_manager.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/mamba_cache_manager.py @@ -3998,8 +3998,9 @@ def _build_cache_config( # The recurrent (SSM) state pool must hold one slot per resident # sequence plus every reserved dummy slot. Unlike attention pages, a # Mamba state is fixed-size per sequence, so this floor is independent - # of sequence length. Speculative decoding inflates the reserved dummy - # slots (CUDA-graph padding), so without this explicit floor the SSM + # of sequence length. The base config only emits constraints when + # ``avg_seq_len`` is set, and speculative decoding inflates the reserved + # dummy slots (CUDA-graph padding), so without an explicit floor the SSM # pool can be undersized (see the live/dummy-slot check in _setup_states # / __init__). Add a min-slots constraint of zero-capacity requests: # these cost no attention pages but reserve one SSM slot each. diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index d4dfcb60d18c..1b4b3d280f39 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -888,7 +888,7 @@ def test_default_uses_allocator_fallback() -> None: assert config.constraints == [] -def test_avg_seq_len_does_not_require_max_length_warmup() -> None: +def test_avg_seq_len_builds_context_warmup_constraint() -> None: config = _make_cache_config_for_test( KvCacheConfig(host_cache_size=0, avg_seq_len=1024), max_batch_size=3, @@ -901,7 +901,7 @@ def test_avg_seq_len_does_not_require_max_length_warmup() -> None: [KVCacheDesc(capacity=2048, history_length=0)] + [KVCacheDesc(capacity=1024, history_length=1021)] * 2 ) - assert config.constraints == [] + assert config.constraints == [BatchDesc([KVCacheDesc(capacity=2048, history_length=0)])] def test_avg_seq_len_updates_typical_step() -> None: @@ -1155,7 +1155,7 @@ def test_extra_tokens_are_in_context_capacity() -> None: ) assert config.typical_step == BatchDesc([KVCacheDesc(capacity=258, history_length=0)]) - assert config.constraints == [] + assert config.constraints == [BatchDesc([KVCacheDesc(capacity=258, history_length=0)])] def test_try_commit_blocks_commits_partial_block_at_context_end() -> None: From 28760885493b95906951e9136446bbda410a76c7 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:13:46 -0700 Subject: [PATCH 12/15] [None][fix] Avoid counting the generation dummy input twice Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../kv_cache/kv_cache_manager_v2.py | 3 +- .../_torch/pyexecutor/model_engine.py | 8 ---- .../tools/layer_wise_benchmarks/runner.py | 3 +- .../kv_cache/test_kvcm2_integration.py | 38 +++++++++++++++++++ 4 files changed, 41 insertions(+), 11 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py index 2e572129eb00..5b6e1b47daac 100644 --- a/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py +++ b/tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py @@ -5158,7 +5158,8 @@ def release_resources( req.total_input_len_cp = token_num * self._helix_cp_size - 1 req.py_decoding_iter = 1 if prepare_resource: - new_capacity = kv_cache.capacity + _kv_draft + 1 + # token_num already includes the current generation input. + new_capacity = kv_cache.capacity + _kv_draft success = kv_cache.resize(new_capacity, history_length=history_hint) if not success: release_resources(req, free_draft_resources=draft_kv_cache is not None) diff --git a/tensorrt_llm/_torch/pyexecutor/model_engine.py b/tensorrt_llm/_torch/pyexecutor/model_engine.py index 7a73d016c8d4..2429877c9cb4 100644 --- a/tensorrt_llm/_torch/pyexecutor/model_engine.py +++ b/tensorrt_llm/_torch/pyexecutor/model_engine.py @@ -3491,14 +3491,6 @@ def free_warmup_requests() -> None: max_num_draft_tokens=_kv_draft) available_tokens = min(available_tokens, draft_available_tokens) - if isinstance(kv_cache_manager, KVCacheManagerV2): - # V2 reserves one generation token beyond the draft/extra tokens. - available_tokens -= 1 - minimum_tokens = ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1 - if available_tokens < minimum_tokens: - free_warmup_requests() - return None - token_num = max( ENC_DEC_CUDA_GRAPH_DUMMY_TOKEN_NUM if is_enc_dec else 1, min( diff --git a/tensorrt_llm/tools/layer_wise_benchmarks/runner.py b/tensorrt_llm/tools/layer_wise_benchmarks/runner.py index c2e58941d132..438e181797f7 100644 --- a/tensorrt_llm/tools/layer_wise_benchmarks/runner.py +++ b/tensorrt_llm/tools/layer_wise_benchmarks/runner.py @@ -902,8 +902,7 @@ def create_kv_cache_manager( # Please refer to `tensorrt_llm/_torch/pyexecutor/_util.py` for `kv_cache_manager` config = model_config.pretrained_config - # max_seq_len + 1 because the is_gen path in add_dummy_requests resizes each - # request to capacity + 1; without the extra token the last block rounds down. + # Reserve one token of headroom before rounding to a block boundary. # kv_pool_headroom oversizes max_tokens when the manager splits it across # several pools. DeepSeek-V4 needs 3; the default 1 keeps every other model # on its previous allocation. diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index 1b4b3d280f39..bcc349de6617 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -1565,6 +1565,44 @@ def test_external_draft_estimated_quota_supports_allocation_and_resume( manager.shutdown() +@pytest.mark.parametrize("draft_len", [0, 4]) +def test_generation_dummy_uses_available_capacity(draft_len: int) -> None: + if not torch.cuda.is_available(): + pytest.skip("requires CUDA") + init_cuda_once() + manager = KVCacheManagerV2( + KvCacheConfig(enable_block_reuse=False, max_gpu_total_bytes=4 << 20), + CacheType.SELF, + num_layers=2, + num_kv_heads=2, + head_dim=128, + tokens_per_block=32, + max_seq_len=131072, + max_batch_size=1, + max_num_tokens=128, + mapping=Mapping(), + dtype=DataType.HALF, + spec_config=MTPDecodingConfig(max_draft_len=draft_len) if draft_len else None, + ) + try: + token_num = manager.get_num_available_tokens( + token_num_upper_bound=manager.max_seq_len, max_num_draft_tokens=draft_len + ) + capacity = token_num + manager.num_extra_kv_tokens + draft_len + assert capacity % manager.tokens_per_block == 0 + # The current generation input is already included in token_num. + requests = manager.add_dummy_requests( + [0], token_nums=[token_num], is_gen=True, max_num_draft_tokens=draft_len + ) + assert requests is not None + cache = manager.kv_cache_map[requests[0].py_request_id] + assert cache.history_length == token_num - 1 + assert cache.capacity == capacity + manager.free_resources(requests[0]) + finally: + manager.shutdown() + + @pytest.fixture def max_num_turns() -> int: return 1 From cf8bd33c73c3c2d1c2df0c2d836fe7f949b84a4b Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:42:43 -0700 Subject: [PATCH 13/15] [None][test] Cover DeepSeek V4 warmup constraint configuration Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../sparse/deepseek_v4/cache_manager.py | 1 + .../test_deepseek_v4_cache_manager.py | 47 +++++++++++++++++++ 2 files changed, 48 insertions(+) diff --git a/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py b/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py index b528380ff0a7..7b9e6dbd3815 100644 --- a/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py +++ b/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py @@ -1216,6 +1216,7 @@ def _add_layer( self._num_manager_layers = len(layers) constraints = list(config.constraints) + # _build_base_config copies kv_cache_config.pool_ratio to initial_pool_ratio. if config.initial_pool_ratio is None: # DeepSeek-V4's windowed and compressed pools must also support # the longest decode request alongside the short decode requests. diff --git a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py index 3d757b10ec38..932d8ae1d8b2 100644 --- a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py +++ b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py @@ -452,6 +452,53 @@ def test_typical_seq_len_preserves_deepseek_v4_fallback( assert manager._get_typical_seq_len(KvCacheConfig(avg_seq_len=avg_seq_len)) == expected +@pytest.mark.cpu_only +@pytest.mark.parametrize("pool_ratio", [None, [1.0]]) +def test_build_cache_config_long_decode_constraint(pool_ratio: list[float] | None) -> None: + manager = object.__new__(DeepseekV4CacheManager) + manager.pp_layers = [0] + manager._compress_ratios = [1] + manager.dtype = DataType.BF16 + manager._use_nvfp4_compress = False + manager.head_dim = 512 + 64 + manager.index_head_dim = 128 + manager._indexer_k_dtype = "fp8" + manager.use_fp8_ds_mla = False + manager._swa_window_size = 128 + manager.tokens_per_block = 128 + manager.max_seq_len = 1024 + manager.max_batch_size = 3 + manager.max_draft_len = manager._max_draft_len = 4 + manager.num_extra_kv_tokens = 3 + + context_constraint = BatchDesc([KVCacheDesc(capacity=259, history_length=0)]) + base_config = KVCacheManagerConfig( + tokens_per_block=128, + cache_tiers=[GpuCacheTierConfig(quota=1 << 20)], + layers=[], + constraints=[context_constraint] if pool_ratio is None else [], + initial_pool_ratio=pool_ratio, + ) + + config = manager._build_cache_config(base_config) + + if pool_ratio is None: + assert config.constraints == [ + context_constraint, + BatchDesc( + [ + KVCacheDesc(capacity=1024, history_length=1023), + KVCacheDesc(capacity=8, history_length=0), + KVCacheDesc(capacity=8, history_length=0), + ] + ), + ] + assert base_config.constraints == [context_constraint] + else: + assert config.initial_pool_ratio == pool_ratio + assert config.constraints == [] + + def test_cache_size_estimation_uses_model_attention_layer_count(): class FakeModelConfig: sparse_attention_config = SimpleNamespace( From 141fc0b09e7b4182ca0de8d2b8d6b4e18e1ffdff Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Thu, 1 Oct 2026 13:08:25 +0000 Subject: [PATCH 14/15] [None][fix] Preserve V4 estimation batch limits and warmup coverage Restore CUDA graph batch clipping when estimating V4's long-decode cache constraint. Keep the generation dummy input counted once and align the allocation tests with that contract, including the actual graph builder. Validate dynamic draft schedules against observed batch sizes while requiring coverage of each configured draft length, without depending on exact request admission thresholds. Validation: 154 cases passed on a B200 using the CI 62214 wheel with the V4 runtime fix overlaid, including 15 attention graph capture/replay cases. Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../sparse/deepseek_v4/cache_manager.py | 11 ++++- .../test_deepseek_v4_cache_manager.py | 43 +++++++++++----- .../kv_cache/test_kvcm2_integration.py | 49 +++++++++++++------ .../hw_agnostic/test_draft_len_schedule.py | 8 +-- 4 files changed, 81 insertions(+), 30 deletions(-) diff --git a/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py b/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py index 7b9e6dbd3815..9879c0e40a4b 100644 --- a/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py +++ b/tensorrt_llm/_torch/attention/backends/sparse/deepseek_v4/cache_manager.py @@ -1220,6 +1220,15 @@ def _add_layer( if config.initial_pool_ratio is None: # DeepSeek-V4's windowed and compressed pools must also support # the longest decode request alongside the short decode requests. + constraint_batch_size = self._get_generation_request_capacity() + if ( + self.max_cuda_graph_batch_size is not None + and self.max_cuda_graph_batch_size > 0 + and self.is_estimating_kv_cache + and all(window is None for window in self.max_attention_window_vec) + ): + constraint_batch_size = min(constraint_batch_size, self.max_cuda_graph_batch_size) + constraint_batch_size = max(1, constraint_batch_size) min_decode_capacity = 1 + self.max_draft_len + self.num_extra_kv_tokens constraints.append( BatchDesc( @@ -1230,7 +1239,7 @@ def _add_layer( ) ] + [KVCacheDesc(capacity=min_decode_capacity, history_length=0)] - * (self.max_batch_size - 1) + * (constraint_batch_size - 1) ) ) return replace( diff --git a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py index 932d8ae1d8b2..95bbd9c9adae 100644 --- a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py +++ b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py @@ -454,7 +454,24 @@ def test_typical_seq_len_preserves_deepseek_v4_fallback( @pytest.mark.cpu_only @pytest.mark.parametrize("pool_ratio", [None, [1.0]]) -def test_build_cache_config_long_decode_constraint(pool_ratio: list[float] | None) -> None: +@pytest.mark.parametrize( + "estimating,graph_batch_size,window,expected_batch_size", + [ + (False, 2, None, 3), + (True, 2, None, 2), + (True, None, None, 3), + (True, 0, None, 3), + (True, 4, None, 3), + (True, 2, 128, 3), + ], +) +def test_build_cache_config_long_decode_constraint( + pool_ratio: list[float] | None, + estimating: bool, + graph_batch_size: int | None, + window: int | None, + expected_batch_size: int, +) -> None: manager = object.__new__(DeepseekV4CacheManager) manager.pp_layers = [0] manager._compress_ratios = [1] @@ -468,6 +485,9 @@ def test_build_cache_config_long_decode_constraint(pool_ratio: list[float] | Non manager.tokens_per_block = 128 manager.max_seq_len = 1024 manager.max_batch_size = 3 + manager.is_estimating_kv_cache = estimating + manager.max_cuda_graph_batch_size = graph_batch_size + manager.max_attention_window_vec = [window] manager.max_draft_len = manager._max_draft_len = 4 manager.num_extra_kv_tokens = 3 @@ -486,11 +506,8 @@ def test_build_cache_config_long_decode_constraint(pool_ratio: list[float] | Non assert config.constraints == [ context_constraint, BatchDesc( - [ - KVCacheDesc(capacity=1024, history_length=1023), - KVCacheDesc(capacity=8, history_length=0), - KVCacheDesc(capacity=8, history_length=0), - ] + [KVCacheDesc(capacity=1024, history_length=1023)] + + [KVCacheDesc(capacity=8, history_length=0)] * (expected_batch_size - 1) ), ] assert base_config.constraints == [context_constraint] @@ -2481,19 +2498,23 @@ def test_disagg_generation_init_rejects_context_apis(self): finally: cache_manager.shutdown() - def test_dummy_generation_requests_with_swa_scratch_reuse(self): + @pytest.mark.parametrize("compress_ratios", [[1], [1, 4, 128]]) + @pytest.mark.parametrize("long_token_num", [127, 128, 129, 513]) + def test_dummy_generation_requests_with_swa_scratch_reuse( + self, compress_ratios: list[int], long_token_num: int + ) -> None: cache_manager, _ = self._create_deepseek_v4_cache_manager( tokens_per_block=self.tokens_per_block, max_batch_size=2, max_seq_len=1024, - compress_ratios=[1], + compress_ratios=compress_ratios, dtype=DataType.BF16, compressor_dtype=DataType.FLOAT, enable_swa_scratch_reuse=True, ) requests = [] - token_nums = [1, self.tokens_per_block * 4 + 1] + token_nums = [1, long_token_num] try: requests = cache_manager.add_dummy_requests( request_ids=[0, 1], @@ -2508,9 +2529,9 @@ def test_dummy_generation_requests_with_swa_scratch_reuse(self): assert not short_kv_cache.enable_swa_scratch_reuse assert not long_kv_cache.enable_swa_scratch_reuse assert short_kv_cache.history_length == 0 - assert short_kv_cache.capacity == token_nums[0] + 1 + assert short_kv_cache.capacity == token_nums[0] assert long_kv_cache.history_length == token_nums[1] - 1 - assert long_kv_cache.capacity == token_nums[1] + 1 + assert long_kv_cache.capacity == token_nums[1] finally: for req in requests: cache_manager.free_resources(req) diff --git a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py index bcc349de6617..8ce906b9921c 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py +++ b/tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py @@ -36,6 +36,8 @@ _update_kv_cache_draft_token_location, ) from tensorrt_llm._torch.pyexecutor.llm_request import LlmRequest, LlmRequestState +from tensorrt_llm._torch.pyexecutor.model_engine import PyTorchModelEngine +from tensorrt_llm._torch.pyexecutor.resource_manager import ResourceManager, ResourceManagerType from tensorrt_llm._torch.pyexecutor.scheduler import ScheduledRequests from tensorrt_llm.bindings import DataType, SamplingConfig from tensorrt_llm.bindings.BuildInfo import ENABLE_MULTI_DEVICE @@ -1569,7 +1571,8 @@ def test_external_draft_estimated_quota_supports_allocation_and_resume( def test_generation_dummy_uses_available_capacity(draft_len: int) -> None: if not torch.cuda.is_available(): pytest.skip("requires CUDA") - init_cuda_once() + torch.cuda.init() + spec_config = MTPDecodingConfig(max_draft_len=draft_len) if draft_len else None manager = KVCacheManagerV2( KvCacheConfig(enable_block_reuse=False, max_gpu_total_bytes=4 << 20), CacheType.SELF, @@ -1578,27 +1581,43 @@ def test_generation_dummy_uses_available_capacity(draft_len: int) -> None: head_dim=128, tokens_per_block=32, max_seq_len=131072, - max_batch_size=1, + max_batch_size=2, max_num_tokens=128, mapping=Mapping(), dtype=DataType.HALF, - spec_config=MTPDecodingConfig(max_draft_len=draft_len) if draft_len else None, + spec_config=spec_config, ) try: - token_num = manager.get_num_available_tokens( - token_num_upper_bound=manager.max_seq_len, max_num_draft_tokens=draft_len + engine = SimpleNamespace( + kv_cache_manager_key=ResourceManagerType.KV_CACHE_MANAGER, + spec_config=spec_config, + max_draft_len=draft_len, + max_draft_loop_tokens=draft_len, + max_seq_len=manager.max_seq_len, + max_beam_width=1, + use_mrope=False, + get_runtime_tokens_per_gen_step=lambda length: length + 1, + _get_draft_kv_cache_manager=lambda _: None, + _is_encoder_decoder_model=lambda: False, + model=SimpleNamespace( + model_config=SimpleNamespace(pretrained_config=SimpleNamespace()) + ), ) - capacity = token_num + manager.num_extra_kv_tokens + draft_len - assert capacity % manager.tokens_per_block == 0 - # The current generation input is already included in token_num. - requests = manager.add_dummy_requests( - [0], token_nums=[token_num], is_gen=True, max_num_draft_tokens=draft_len + resources = ResourceManager({ResourceManagerType.KV_CACHE_MANAGER: manager}) + batch = PyTorchModelEngine._create_cuda_graph_warmup_request( + engine, resources, batch_size=2, draft_len=draft_len ) - assert requests is not None - cache = manager.kv_cache_map[requests[0].py_request_id] - assert cache.history_length == token_num - 1 - assert cache.capacity == capacity - manager.free_resources(requests[0]) + assert batch is not None + assert len(batch.generation_requests) == 2 + longest_cache = manager.kv_cache_map[batch.generation_requests[0].py_request_id] + assert longest_cache.capacity % manager.tokens_per_block == 0 + for request in batch.generation_requests: + cache = manager.kv_cache_map[request.py_request_id] + token_num = request.prompt_len + 1 + assert cache.history_length == request.prompt_len + assert cache.capacity == token_num + manager.num_extra_kv_tokens + draft_len + manager.free_resources(request) + assert not manager.kv_cache_map finally: manager.shutdown() diff --git a/tests/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py b/tests/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py index 124026bcb521..679162e7f3ac 100644 --- a/tests/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py +++ b/tests/unittest/_torch/speculative/hw_agnostic/test_draft_len_schedule.py @@ -166,9 +166,11 @@ def instrumented_handle_dynamic_draft_len(scheduled_batch): for index, observation in enumerate(runtime_schedule) if index == 0 or observation != runtime_schedule[index - 1] ] - expected_key_transitions = {(8, 1), (4, 2), (1, 3)} - assert expected_key_transitions.issubset(runtime_transitions), ( - f"DraftTarget runtime schedule did not exercise {expected_key_transitions}: " + # Request admission and completion may skip exact batch-size thresholds. + expected_draft_lengths = set(schedule.values()) + observed_draft_lengths = {draft_len for _, draft_len in runtime_schedule} + assert expected_draft_lengths.issubset(observed_draft_lengths), ( + f"DraftTarget runtime schedule did not exercise draft lengths {expected_draft_lengths}: " f"got {runtime_transitions}" ) return From 45fc4ac5c9720adf0bb6bb3f892c8fdda9c5ed31 Mon Sep 17 00:00:00 2001 From: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> Date: Sun, 4 Oct 2026 04:31:19 +0000 Subject: [PATCH 15/15] [None][test] Complete V4 sparse offload cache fixture initialization Signed-off-by: Yi Zhang <187001205+yizhang-nv@users.noreply.github.com> --- .../sparse/deepseek_v4/test_deepseek_v4_cache_manager.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py index 95bbd9c9adae..2dbcd2c472c7 100644 --- a/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py +++ b/tests/unittest/_torch/attention/sparse/deepseek_v4/test_deepseek_v4_cache_manager.py @@ -370,7 +370,13 @@ def test_sparse_offload_cache_roles( manager._use_nvfp4_compress = False manager.use_fp8_ds_mla = False manager._swa_window_size = 128 - manager._max_draft_len = 0 + manager.max_seq_len = 1024 + manager.max_batch_size = 2 + manager.is_estimating_kv_cache = False + manager.max_cuda_graph_batch_size = None + manager.max_attention_window_vec = [None] * len(pp_layers) + manager.max_draft_len = manager._max_draft_len = 0 + manager.num_extra_kv_tokens = 0 config = KVCacheManagerConfig( tokens_per_block=tokens_per_block, cache_tiers=[