Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
15 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -47,10 +47,12 @@
from tensorrt_llm.runtime.kv_cache_manager_v2 import (
BAD_PAGE_INDEX,
AttentionLayerConfig,
BatchDesc,
BufferConfig,
BufferId,
CudaStream,
DataRole,
KVCacheDesc,
LayerId,
PageIndexMode,
ScratchDesc,
Expand Down Expand Up @@ -1062,7 +1064,7 @@ def _get_max_tokens_from_quota(self, quota: int) -> float:

def _build_cache_config(self, config: KVCacheManagerConfigPy) -> KVCacheManagerConfigPy:
"""
Add DeepSeek-V4 layers to the cache config.
Add DeepSeek-V4 layers and warmup constraints to the cache config.
"""
layers: List[AttentionLayerConfig] = []
layer_attn_to_layer_id: Dict[Tuple[int, DeepseekV4AttentionType], LayerId] = {}
Expand Down Expand Up @@ -1213,9 +1215,37 @@ def _add_layer(
# number of layers in the KVCacheManagerPy
self._num_manager_layers = len(layers)

constraints = list(config.constraints)

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This constraint has no unit coverage — test_deepseek_v4_cache_manager.py never references constraints or _build_cache_config, and the generic V2 test now only asserts its absence. Since this is the second time the constraint has moved, please add a CPU-only test that builds the V4 config and asserts the emitted BatchDesc (one max_seq_len decode plus max_batch_size - 1 minimal decodes, with 1 + max_draft_len + num_extra_kv_tokens capacity), plus a case with pool_ratio set asserting it is omitted.

Separately: the gate here is config.initial_pool_ratio is None while the base gates the same block on kv_cache_config.pool_ratio is None. They agree today only because _build_base_config assigns initial_pool_ratio=kv_cache_config.pool_ratio. A one-line comment noting that coupling would save the next reader the trip.

# _build_base_config copies kv_cache_config.pool_ratio to initial_pool_ratio.
if config.initial_pool_ratio is None:
# DeepSeek-V4's windowed and compressed pools must also support
# the longest decode request alongside the short decode requests.
constraint_batch_size = self._get_generation_request_capacity()
if (
self.max_cuda_graph_batch_size is not None
and self.max_cuda_graph_batch_size > 0
and self.is_estimating_kv_cache
and all(window is None for window in self.max_attention_window_vec)
):
constraint_batch_size = min(constraint_batch_size, self.max_cuda_graph_batch_size)
constraint_batch_size = max(1, constraint_batch_size)
min_decode_capacity = 1 + self.max_draft_len + self.num_extra_kv_tokens
constraints.append(
BatchDesc(
[
KVCacheDesc(
capacity=self.max_seq_len,
history_length=self.max_seq_len - 1,
)
]
+ [KVCacheDesc(capacity=min_decode_capacity, history_length=0)]
* (constraint_batch_size - 1)
)
)
return replace(
config,
layers=layers,
constraints=constraints,
)

def _init_indexer_dtype(self, sparse_attn_config: DeepSeekV4SparseAttentionConfig) -> None:
Expand Down
35 changes: 2 additions & 33 deletions tensorrt_llm/_torch/pyexecutor/kv_cache/kv_cache_manager_v2.py
Original file line number Diff line number Diff line change
Expand Up @@ -2798,38 +2798,6 @@ def _build_base_config(
* (generation_request_capacity - 1)
)

# CUDA graph generation warmup uses one request at max_seq_len and
# enough minimal decode requests to fill the resident capacity.
if (
self.max_cuda_graph_batch_size is not None
and self.max_cuda_graph_batch_size > 0
and self.is_estimating_kv_cache
and all(window is None for window in self.max_attention_window_vec)
):
# Estimation graph warmup needs the smaller of the resident
# capacity and the largest captured CUDA graph batch.
constraint_batch_size = min(
generation_request_capacity, self.max_cuda_graph_batch_size
)
else:
constraint_batch_size = generation_request_capacity
constraint_batch_size = max(1, constraint_batch_size)
min_decode_capacity = 1 + self.max_draft_len + self.num_extra_kv_tokens
# Model one request at max_seq_len plus minimal decode requests
# to fill constraint_batch_size.
constraints.append(
BatchDesc(
[
KVCacheDesc(
capacity=self.max_seq_len,
history_length=self.max_seq_len - 1,
)
]
+ [KVCacheDesc(capacity=min_decode_capacity, history_length=0)]
* (constraint_batch_size - 1)
)
)

# General and chunked-prefill warmup uses one fresh context request
# at the per-iteration token budget.
if self.max_num_tokens is not None:
Expand Down Expand Up @@ -5190,7 +5158,8 @@ def release_resources(
req.total_input_len_cp = token_num * self._helix_cp_size - 1
req.py_decoding_iter = 1
if prepare_resource:
new_capacity = kv_cache.capacity + _kv_draft + 1
# token_num already includes the current generation input.
new_capacity = kv_cache.capacity + _kv_draft
success = kv_cache.resize(new_capacity, history_length=history_hint)
if not success:
release_resources(req, free_draft_resources=draft_kv_cache is not None)
Expand Down
3 changes: 1 addition & 2 deletions tensorrt_llm/tools/layer_wise_benchmarks/runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -902,8 +902,7 @@ def create_kv_cache_manager(

# Please refer to `tensorrt_llm/_torch/pyexecutor/_util.py` for `kv_cache_manager`
config = model_config.pretrained_config
# max_seq_len + 1 because the is_gen path in add_dummy_requests resizes each
# request to capacity + 1; without the extra token the last block rounds down.
# Reserve one token of headroom before rounding to a block boundary.
# kv_pool_headroom oversizes max_tokens when the manager splits it across
# several pools. DeepSeek-V4 needs 3; the default 1 keeps every other model
# on its previous allocation.
Expand Down
3 changes: 2 additions & 1 deletion tests/integration/defs/accuracy/test_llm_api_pytorch.py
Original file line number Diff line number Diff line change
Expand Up @@ -6126,7 +6126,8 @@ class TestSeedOss_36B(LlmapiAccuracyTestHarness):
@pytest.mark.timeout(14400)
@pytest.mark.skip_less_device_memory(140000)
def test_auto_dtype(self):
kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.8)
kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.8,

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

SeedOssForCausalLM doesn't declare get_preferred_kv_cache_manager_version, so use_kv_cache_manager_v2="auto" resolves to V1 today — this pins a 4-hour GSM8K run to V2 and drops the default-path coverage it had. Same for the Mistral, gRPC VLM, and multimodal-example fixtures.

If the intent is a regression guard for this fix, parametrizing over [False, True] (or adding a separate V2 case) keeps both paths covered. If the intent is that V2 is where these models are headed, say so in the PR description so the coverage trade is deliberate rather than incidental.

use_kv_cache_manager_v2=True)
chat_template_kwargs = dict(thinking_budget=-1)

with LLM(self.MODEL_PATH, kv_cache_config=kv_cache_config) as llm:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -639,7 +639,7 @@ class TestMistralSmall24B(LlmapiAccuracyTestHarness):
ids=["forced_chunked_prefill"],
)
def test_auto_dtype(self, max_num_tokens):
kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.75)
kv_cache_config = KvCacheConfig(free_gpu_memory_fraction=0.75, use_kv_cache_manager_v2=True)
with LLM(
self.MODEL_PATH,
kv_cache_config=kv_cache_config,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -370,7 +370,13 @@ def test_sparse_offload_cache_roles(
manager._use_nvfp4_compress = False
manager.use_fp8_ds_mla = False
manager._swa_window_size = 128
manager._max_draft_len = 0
manager.max_seq_len = 1024
manager.max_batch_size = 2
manager.is_estimating_kv_cache = False
manager.max_cuda_graph_batch_size = None
manager.max_attention_window_vec = [None] * len(pp_layers)
manager.max_draft_len = manager._max_draft_len = 0
manager.num_extra_kv_tokens = 0
config = KVCacheManagerConfig(
tokens_per_block=tokens_per_block,
cache_tiers=[
Expand Down Expand Up @@ -452,6 +458,70 @@ def test_typical_seq_len_preserves_deepseek_v4_fallback(
assert manager._get_typical_seq_len(KvCacheConfig(avg_seq_len=avg_seq_len)) == expected


@pytest.mark.cpu_only
@pytest.mark.parametrize("pool_ratio", [None, [1.0]])
@pytest.mark.parametrize(
"estimating,graph_batch_size,window,expected_batch_size",
[
(False, 2, None, 3),
(True, 2, None, 2),
(True, None, None, 3),
(True, 0, None, 3),
(True, 4, None, 3),
(True, 2, 128, 3),
],
)
def test_build_cache_config_long_decode_constraint(
pool_ratio: list[float] | None,
estimating: bool,
graph_batch_size: int | None,
window: int | None,
expected_batch_size: int,
) -> None:
manager = object.__new__(DeepseekV4CacheManager)
manager.pp_layers = [0]
manager._compress_ratios = [1]
manager.dtype = DataType.BF16
manager._use_nvfp4_compress = False
manager.head_dim = 512 + 64
manager.index_head_dim = 128
manager._indexer_k_dtype = "fp8"
manager.use_fp8_ds_mla = False
manager._swa_window_size = 128
manager.tokens_per_block = 128
manager.max_seq_len = 1024
manager.max_batch_size = 3
manager.is_estimating_kv_cache = estimating
manager.max_cuda_graph_batch_size = graph_batch_size
manager.max_attention_window_vec = [window]
manager.max_draft_len = manager._max_draft_len = 4
manager.num_extra_kv_tokens = 3

context_constraint = BatchDesc([KVCacheDesc(capacity=259, history_length=0)])
base_config = KVCacheManagerConfig(
tokens_per_block=128,
cache_tiers=[GpuCacheTierConfig(quota=1 << 20)],
layers=[],
constraints=[context_constraint] if pool_ratio is None else [],
initial_pool_ratio=pool_ratio,
)

config = manager._build_cache_config(base_config)

if pool_ratio is None:
assert config.constraints == [
context_constraint,
BatchDesc(
[KVCacheDesc(capacity=1024, history_length=1023)]
+ [KVCacheDesc(capacity=8, history_length=0)] * (expected_batch_size - 1)
),
]
assert base_config.constraints == [context_constraint]
else:
assert config.initial_pool_ratio == pool_ratio
assert config.constraints == []


def test_cache_size_estimation_uses_model_attention_layer_count():
class FakeModelConfig:
sparse_attention_config = SimpleNamespace(
Expand Down Expand Up @@ -2434,19 +2504,23 @@ def test_disagg_generation_init_rejects_context_apis(self):
finally:
cache_manager.shutdown()

def test_dummy_generation_requests_with_swa_scratch_reuse(self):
@pytest.mark.parametrize("compress_ratios", [[1], [1, 4, 128]])
@pytest.mark.parametrize("long_token_num", [127, 128, 129, 513])
def test_dummy_generation_requests_with_swa_scratch_reuse(
self, compress_ratios: list[int], long_token_num: int
) -> None:
cache_manager, _ = self._create_deepseek_v4_cache_manager(
tokens_per_block=self.tokens_per_block,
max_batch_size=2,
max_seq_len=1024,
compress_ratios=[1],
compress_ratios=compress_ratios,
dtype=DataType.BF16,
compressor_dtype=DataType.FLOAT,
enable_swa_scratch_reuse=True,
)

requests = []
token_nums = [1, self.tokens_per_block * 4 + 1]
token_nums = [1, long_token_num]
try:
requests = cache_manager.add_dummy_requests(
request_ids=[0, 1],
Expand All @@ -2461,9 +2535,9 @@ def test_dummy_generation_requests_with_swa_scratch_reuse(self):
assert not short_kv_cache.enable_swa_scratch_reuse
assert not long_kv_cache.enable_swa_scratch_reuse
assert short_kv_cache.history_length == 0
assert short_kv_cache.capacity == token_nums[0] + 1
assert short_kv_cache.capacity == token_nums[0]
assert long_kv_cache.history_length == token_nums[1] - 1
assert long_kv_cache.capacity == token_nums[1] + 1
assert long_kv_cache.capacity == token_nums[1]
finally:
for req in requests:
cache_manager.free_resources(req)
Expand Down
72 changes: 60 additions & 12 deletions tests/unittest/_torch/executor/kv_cache/test_kvcm2_integration.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,8 @@
_update_kv_cache_draft_token_location,
)
from tensorrt_llm._torch.pyexecutor.llm_request import LlmRequest, LlmRequestState
from tensorrt_llm._torch.pyexecutor.model_engine import PyTorchModelEngine
from tensorrt_llm._torch.pyexecutor.resource_manager import ResourceManager, ResourceManagerType
from tensorrt_llm._torch.pyexecutor.scheduler import ScheduledRequests
from tensorrt_llm.bindings import DataType, SamplingConfig
from tensorrt_llm.bindings.BuildInfo import ENABLE_MULTI_DEVICE
Expand Down Expand Up @@ -888,7 +890,7 @@ def test_default_uses_allocator_fallback() -> None:
assert config.constraints == []


def test_avg_seq_len_builds_warmup_constraints() -> None:
def test_avg_seq_len_builds_context_warmup_constraint() -> None:
config = _make_cache_config_for_test(
KvCacheConfig(host_cache_size=0, avg_seq_len=1024),
max_batch_size=3,
Expand All @@ -901,16 +903,7 @@ def test_avg_seq_len_builds_warmup_constraints() -> None:
[KVCacheDesc(capacity=2048, history_length=0)]
+ [KVCacheDesc(capacity=1024, history_length=1021)] * 2
)
assert config.constraints == [
BatchDesc(
[
KVCacheDesc(capacity=1024, history_length=1023),
KVCacheDesc(capacity=3, history_length=0),
KVCacheDesc(capacity=3, history_length=0),
]
),
BatchDesc([KVCacheDesc(capacity=2048, history_length=0)]),
]
assert config.constraints == [BatchDesc([KVCacheDesc(capacity=2048, history_length=0)])]


def test_avg_seq_len_updates_typical_step() -> None:
Expand Down Expand Up @@ -1164,7 +1157,7 @@ def test_extra_tokens_are_in_context_capacity() -> None:
)

assert config.typical_step == BatchDesc([KVCacheDesc(capacity=258, history_length=0)])
assert config.constraints[1] == BatchDesc([KVCacheDesc(capacity=258, history_length=0)])
assert config.constraints == [BatchDesc([KVCacheDesc(capacity=258, history_length=0)])]


def test_try_commit_blocks_commits_partial_block_at_context_end() -> None:
Expand Down Expand Up @@ -1574,6 +1567,61 @@ def test_external_draft_estimated_quota_supports_allocation_and_resume(
manager.shutdown()


@pytest.mark.parametrize("draft_len", [0, 4])
def test_generation_dummy_uses_available_capacity(draft_len: int) -> None:
if not torch.cuda.is_available():
pytest.skip("requires CUDA")
torch.cuda.init()
spec_config = MTPDecodingConfig(max_draft_len=draft_len) if draft_len else None
manager = KVCacheManagerV2(
KvCacheConfig(enable_block_reuse=False, max_gpu_total_bytes=4 << 20),
CacheType.SELF,
num_layers=2,
num_kv_heads=2,
head_dim=128,
tokens_per_block=32,
max_seq_len=131072,
max_batch_size=2,
max_num_tokens=128,
mapping=Mapping(),
dtype=DataType.HALF,
spec_config=spec_config,
)
try:
engine = SimpleNamespace(
kv_cache_manager_key=ResourceManagerType.KV_CACHE_MANAGER,
spec_config=spec_config,
max_draft_len=draft_len,
max_draft_loop_tokens=draft_len,
max_seq_len=manager.max_seq_len,
max_beam_width=1,
use_mrope=False,
get_runtime_tokens_per_gen_step=lambda length: length + 1,
_get_draft_kv_cache_manager=lambda _: None,
_is_encoder_decoder_model=lambda: False,
model=SimpleNamespace(
model_config=SimpleNamespace(pretrained_config=SimpleNamespace())
),
)
resources = ResourceManager({ResourceManagerType.KV_CACHE_MANAGER: manager})
batch = PyTorchModelEngine._create_cuda_graph_warmup_request(
engine, resources, batch_size=2, draft_len=draft_len
)
assert batch is not None
assert len(batch.generation_requests) == 2
longest_cache = manager.kv_cache_map[batch.generation_requests[0].py_request_id]
assert longest_cache.capacity % manager.tokens_per_block == 0
for request in batch.generation_requests:
cache = manager.kv_cache_map[request.py_request_id]
token_num = request.prompt_len + 1
assert cache.history_length == request.prompt_len
assert cache.capacity == token_num + manager.num_extra_kv_tokens + draft_len
manager.free_resources(request)
assert not manager.kv_cache_map
finally:
manager.shutdown()


@pytest.fixture
def max_num_turns() -> int:
return 1
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -166,9 +166,11 @@ def instrumented_handle_dynamic_draft_len(scheduled_batch):
for index, observation in enumerate(runtime_schedule)
if index == 0 or observation != runtime_schedule[index - 1]
]
expected_key_transitions = {(8, 1), (4, 2), (1, 3)}
assert expected_key_transitions.issubset(runtime_transitions), (
f"DraftTarget runtime schedule did not exercise {expected_key_transitions}: "
# Request admission and completion may skip exact batch-size thresholds.
expected_draft_lengths = set(schedule.values())
observed_draft_lengths = {draft_len for _, draft_len in runtime_schedule}
assert expected_draft_lengths.issubset(observed_draft_lengths), (
f"DraftTarget runtime schedule did not exercise draft lengths {expected_draft_lengths}: "
f"got {runtime_transitions}"
)
return
Expand Down
2 changes: 1 addition & 1 deletion tests/unittest/grpc/smg/test_smg.py
Original file line number Diff line number Diff line change
Expand Up @@ -842,7 +842,7 @@ def grpc_vlm_service():
model_path = get_model_path(vlm_model_name)
llm = LLM(
model=model_path,
kv_cache_config=KvCacheConfig(free_gpu_memory_fraction=0.6),
kv_cache_config=KvCacheConfig(free_gpu_memory_fraction=0.6, use_kv_cache_manager_v2=True),
load_format="dummy",
)
tokenizer = llm.tokenizer
Expand Down
Loading
Loading