From 047b7eba2130fe396efbdd0d274d22bd710db709 Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Thu, 10 Sep 2026 11:46:52 -0700 Subject: [PATCH 1/9] [https://nvbugs/6713231][fix] Guard cross-KV prefix reuse for feature-driven encoder requests Whisper requests carry encoder features, not encoder token ids, so the request has no encoder unique tokens; with block reuse on (the default) the C++ capacity scheduler dereferenced that empty optional and every generation failed with "bad optional access". Skip the cross-reuse lookup without encoder tokens, disable cross-pool reuse for such models, and cover the default-reuse path with a Whisper integration test. Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- .../batch_manager/capacityScheduler.cpp | 24 ++++++++++++--- tensorrt_llm/_torch/pyexecutor/_util.py | 16 ++++++++++ tensorrt_llm/inputs/registry.py | 12 ++++++++ .../llmapi/test_llm_api_pytorch_whisper.py | 30 ++++++++++++++++++- .../test_lists/test-db/l0_h100.yml | 1 + .../test_lists/test-db/l0_l40s.yml | 1 + 6 files changed, 79 insertions(+), 5 deletions(-) diff --git a/cpp/tensorrt_llm/batch_manager/capacityScheduler.cpp b/cpp/tensorrt_llm/batch_manager/capacityScheduler.cpp index 54f45259d3c8..518ba9e80334 100644 --- a/cpp/tensorrt_llm/batch_manager/capacityScheduler.cpp +++ b/cpp/tensorrt_llm/batch_manager/capacityScheduler.cpp @@ -37,6 +37,14 @@ using kv_cache_manager::BlockKeyHasher; namespace { +/// Feature-driven encoders (e.g. Whisper) carry no encoder token ids, so the request has no +/// encoder unique tokens to key cross-KV blocks on and cross prefix reuse cannot apply to it. +bool hasEncoderUniqueTokens(LlmRequest const& req) +{ + auto const& encoderUniqueTokens = req.getEncoderUniqueTokens(); + return encoderUniqueTokens.has_value() && encoderUniqueTokens.value() != nullptr; +} + std::tuple, std::unordered_set> prefillWithChunkedContextsAlreadyExecuting(RequestList const& activeRequests, kv_cache_manager::BaseKVCacheManager const& kvCacheManager, @@ -59,7 +67,7 @@ prefillWithChunkedContextsAlreadyExecuting(RequestList const& activeRequests, newlyContributedContextBlocks.insert(summary.firstNewBlock.value()); } } - if (crossKvCacheManager && crossKvCacheManager->isEnableBlockReuse()) + if (crossKvCacheManager && crossKvCacheManager->isEnableBlockReuse() && hasEncoderUniqueTokens(*req)) { auto uniqueTokens = *(req->getEncoderUniqueTokens().value()); auto summary = crossKvCacheManager->analyzePrefixReuse(uniqueTokens, *req); @@ -344,12 +352,20 @@ std::tuple GuaranteedNoEvictScheduler::impl( if (crossKvCacheManager && crossKvCacheManager->isEnableBlockReuse() && !crossKvCacheManager->getBlockManager().isVariableWindow()) { - auto uniqueTokens = *(req->getEncoderUniqueTokens().value()); - crossSummary = crossKvCacheManager->analyzePrefixReuse(uniqueTokens, *req); + if (hasEncoderUniqueTokens(*req)) + { + auto uniqueTokens = *(req->getEncoderUniqueTokens().value()); + crossSummary = crossKvCacheManager->analyzePrefixReuse(uniqueTokens, *req); + } + else + { + // Nothing to look up: an empty summary means "no reusable cross blocks". + crossSummary = kv_cache_manager::PrefixReuseSummary{}; + } } } else if (isEncoderInit && crossKvCacheManager && crossKvCacheManager->isEnableBlockReuse() - && !crossKvCacheManager->getBlockManager().isVariableWindow()) + && !crossKvCacheManager->getBlockManager().isVariableWindow() && hasEncoderUniqueTokens(*req)) { // Encoder admission only needs the cross summary for reuse ordering. auto uniqueTokens = *(req->getEncoderUniqueTokens().value()); diff --git a/tensorrt_llm/_torch/pyexecutor/_util.py b/tensorrt_llm/_torch/pyexecutor/_util.py index 50576b75e62b..7e417a292a09 100644 --- a/tensorrt_llm/_torch/pyexecutor/_util.py +++ b/tensorrt_llm/_torch/pyexecutor/_util.py @@ -25,6 +25,8 @@ is_sm_100f, prefer_pinned, str_dtype_to_binding, torch_dtype_to_str) from tensorrt_llm.inputs.multimodal import MultimodalParams +from tensorrt_llm.inputs.registry import \ + input_processor_requires_encoder_features # isort: off from tensorrt_llm.llmapi.llm_args import ( @@ -1962,6 +1964,10 @@ def _split_kv_cache_budget_for_draft( def _is_encoder_decoder(self) -> bool: return self._model_engine.model.model_config.is_encoder_decoder + def _encoder_input_is_features(self) -> bool: + return input_processor_requires_encoder_features( + type(self._model_engine.model)) + @staticmethod def _get_config_int_attr(config, names: tuple[str, ...]) -> Optional[int]: for name in names: @@ -2089,6 +2095,16 @@ def _split_kv_cache_budget_for_cross( self_kv_cache_config = base_kv_cache_config.model_copy() cross_kv_cache_config = base_kv_cache_config.model_copy() + if (cross_kv_cache_config.enable_block_reuse + and self._encoder_input_is_features()): + # Cross blocks are keyed on encoder token ids; a feature-driven + # encoder has none, so reuse can never hit and the C++ scheduler + # must not be asked for a cross prefix-reuse summary. + logger.info( + "Disabling block reuse for the cross-KV cache: the encoder " + "takes feature tensors, so requests carry no encoder tokens " + "to key cross blocks on.") + cross_kv_cache_config.enable_block_reuse = False split_any_budget = False free_fraction = base_kv_cache_config.free_gpu_memory_fraction diff --git a/tensorrt_llm/inputs/registry.py b/tensorrt_llm/inputs/registry.py index 0963bd0367df..ae5d2c8bdad3 100644 --- a/tensorrt_llm/inputs/registry.py +++ b/tensorrt_llm/inputs/registry.py @@ -1061,6 +1061,18 @@ def support_multimodal_disaggregated(model_cls: Type[nn.Module]): return model_cls +def input_processor_requires_encoder_features( + model_cls: Type[nn.Module]) -> bool: + """Whether ``model_cls``'s input processor feeds the encoder a feature tensor. + + Feature-driven encoders (e.g. Whisper's audio encoder) take no encoder + token ids, so such requests have nothing to key cross-KV blocks on. + """ + processor_cls = INPUT_PROCESSOR_REGISTRY._input_processors_cls_by_model_type.get( + model_cls) + return bool(getattr(processor_cls, "requires_encoder_features", False)) + + def register_input_processor( processor_cls: Type[InputProcessor], model_type: str, diff --git a/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py b/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py index 8fd9e5f5855b..2dc76d548648 100644 --- a/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py +++ b/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py @@ -116,6 +116,7 @@ def _make_llm( cuda_graph_batch_sizes: list[int] | None = None, tensor_parallel_size: int = 1, encoder_graphs: bool = False, + enable_block_reuse: bool = False, ) -> LLM: """Build a Whisper LLM for the test matrix, optionally with encoder CUDA graphs.""" # CudaGraphConfig captures the decode step; the enc-dec encoder step opts in @@ -154,7 +155,7 @@ def _make_llm( disable_overlap_scheduler=True, # overlap scheduler unsupported enable_chunked_prefill=False, kv_cache_config=KvCacheConfig( - enable_block_reuse=False, + enable_block_reuse=enable_block_reuse, free_gpu_memory_fraction=_FREE_GPU_MEMORY_FRACTION, cross_kv_cache_fraction=_CROSS_KV_CACHE_FRACTION, use_kv_cache_manager_v2=use_kv_cache_manager_v2, @@ -248,6 +249,33 @@ def test_whisper_pytorch_transcribe_end_to_end(monkeypatch): ] +def test_whisper_pytorch_block_reuse_enabled(monkeypatch): + """Greedy transcription with KV block reuse left at its default (enabled). + + Whisper requests carry encoder features, not encoder token ids, so the + cross-KV pool has nothing to key reuse on; the executor must still admit + and run them (https://nvbugs/6713231). Batch 2 co-schedules two + encoder-init requests, the shape that reached the unguarded cross-reuse + lookup in the C++ capacity scheduler. + """ + monkeypatch.setenv("TLLM_WORKER_USE_SINGLE_PROCESS", "1") + + model_path = _get_whisper_model_path() + wave, sample_rate = soundfile.read(_get_audio_path()) + sampling_params = SamplingParams(temperature=0.0, max_tokens=_MAX_NEW_TOKENS) + + with _make_llm(model_path, enable_block_reuse=True) as llm: + for batch_size in (1, 2): + outputs = llm.generate( + [_audio_prompt(wave, sample_rate) for _ in range(batch_size)], + sampling_params, + ) + for output in outputs: + completion = output.outputs[0] + assert list(completion.token_ids) == _EXPECTED_GREEDY_OUTPUT_TOKEN_IDS + assert _EXPECTED_TRANSCRIPT_FRAGMENT in completion.text.lower() + + @pytest.mark.parametrize("torch_dtype,cuda_graph_batch_sizes,graphs_captured", _BEAM_SEARCH_CASES) def test_whisper_pytorch_beam_search( monkeypatch, torch_dtype, cuda_graph_batch_sizes, graphs_captured diff --git a/tests/integration/test_lists/test-db/l0_h100.yml b/tests/integration/test_lists/test-db/l0_h100.yml index d0ac04317b09..1c5b99ff8072 100644 --- a/tests/integration/test_lists/test-db/l0_h100.yml +++ b/tests/integration/test_lists/test-db/l0_h100.yml @@ -193,6 +193,7 @@ l0_h100: - llmapi/test_llm_api_pytorch_t5.py::test_t5_pytorch_generate_encoder_decoder_end_to_end[bf16-kv-v1-cuda-graph-on-beam2-t5-small] - llmapi/test_llm_api_pytorch_t5.py::test_t5_pytorch_generate_encoder_decoder_end_to_end[bf16-kv-v1-cuda-graph-on-greedy-overlap-t5-small] - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_transcribe_end_to_end + - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_block_reuse_enabled - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_feature_combinations[bf16-kv-v2-decoder-graphs-on-greedy] - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_beam_search[bf16-kv-v1-decoder-graphs-on-beam2] - test_e2e.py::test_trtllm_bench_iteration_log[PyTorch-streaming-meta-llama/Llama-3.1-8B-llama-3.1-model/Meta-Llama-3.1-8B] diff --git a/tests/integration/test_lists/test-db/l0_l40s.yml b/tests/integration/test_lists/test-db/l0_l40s.yml index 05361d0813f7..aa70fa119e8a 100644 --- a/tests/integration/test_lists/test-db/l0_l40s.yml +++ b/tests/integration/test_lists/test-db/l0_l40s.yml @@ -51,6 +51,7 @@ l0_l40s: # The encoder-graphs case also exercises decoder graphs, so it stands in for # a decoder-only case rather than adding to it; KV-v2 stays covered on H100. - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_transcribe_end_to_end + - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_block_reuse_enabled - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_feature_combinations[bf16-kv-v1-encoder-graphs-on-greedy] - condition: ranges: From 3e80cd1d357d4ba0a7c024f3e5f7a081b775783c Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Wed, 16 Sep 2026 14:33:12 -0700 Subject: [PATCH 2/9] [https://nvbugs/6713231][test] Cover feature encoder cross-KV scheduling Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- .../batch_manager/capacitySchedulerTest.cpp | 57 +++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/cpp/tests/unit_tests/batch_manager/capacitySchedulerTest.cpp b/cpp/tests/unit_tests/batch_manager/capacitySchedulerTest.cpp index 804a7f2106f1..9c4689da4943 100644 --- a/cpp/tests/unit_tests/batch_manager/capacitySchedulerTest.cpp +++ b/cpp/tests/unit_tests/batch_manager/capacitySchedulerTest.cpp @@ -2608,6 +2608,21 @@ std::shared_ptr createEncoderInitRequest( EXPECT_EQ(req->getState(), LlmRequestState::kENCODER_INIT); return req; } + +std::shared_ptr createFeatureEncoderInitRequest( + int32_t promptLen, int32_t maxNewTokens, int32_t encoderInputLen, int32_t encoderOutputLen, uint64_t reqId) +{ + auto inputTokens = VecTokens(promptLen, 1); + auto executorReq = tensorrt_llm::executor::Request(inputTokens, maxNewTokens); + auto encoderInputFeatures = tensorrt_llm::executor::Tensor::cpu( + tensorrt_llm::executor::DataType::kFP32, {encoderInputLen, /*featureSize=*/1}); + executorReq.setEncoderInputFeatures(std::move(encoderInputFeatures)); + executorReq.setEncoderOutputLength(encoderOutputLen); + auto req = std::make_shared(reqId, executorReq); + EXPECT_EQ(req->getState(), LlmRequestState::kENCODER_INIT); + EXPECT_FALSE(req->getEncoderUniqueTokens().has_value()); + return req; +} } // namespace // GuaranteedNoEvict: a single encoder-init request is admitted without @@ -2688,3 +2703,45 @@ TEST_F(CapacitySchedulerTest, EncoderInitDoesNotConsumeCrossPool) EXPECT_EQ(crossKvCacheManager->getNumFreeBlocks(), crossFreeBefore) << "policy=" << static_cast(policy); } } + +TEST_F(CapacitySchedulerTest, FeatureEncoderWithoutTokensSkipsCrossPoolReuseAnalysis) +{ + SizeType32 const maxNumRequests = 1; + SizeType32 const tokensPerBlock = 10; + SizeType32 const selfMaxTokens = 100; + SizeType32 const crossMaxTokens = 20; + int32_t const promptLen = 10; + int32_t const encoderInputLen = 20; + int32_t const encoderOutputLen = 20; + + auto kvCacheManager = getKvCacheManager(maxNumRequests, tokensPerBlock, selfMaxTokens, selfMaxTokens, + /*sinkTokenLength=*/0, /*enableReuse=*/true); + auto crossKvCacheManager = getKvCacheManager(maxNumRequests, tokensPerBlock, crossMaxTokens, crossMaxTokens, + /*sinkTokenLength=*/0, /*enableReuse=*/true, kv_cache_manager::CacheType::kCROSS); + auto peftCacheManager = getPeftCacheManager(); + auto capacityScheduler + = CapacityScheduler(maxNumRequests, CapacitySchedulerPolicy::kGUARANTEED_NO_EVICT, kvCacheManager != nullptr, + /*twoStepsLookAhead=*/false, LlmRequestState::kENCODER_INIT, LlmRequestState::kGENERATION_COMPLETE); + auto req = createFeatureEncoderInitRequest( + promptLen, /*maxNewTokens=*/40, encoderInputLen, encoderOutputLen, /*reqId=*/1); + RequestList activeRequests{req}; + + auto expectRequestScheduled = [&]() + { + auto [fittingRequests, fittingDisaggGenInitRequests, pausedRequests] + = capacityScheduler(activeRequests, kvCacheManager, peftCacheManager, crossKvCacheManager); + ASSERT_EQ(fittingRequests.size(), 1u); + EXPECT_EQ(fittingRequests.front()->mRequestId, req->mRequestId); + EXPECT_TRUE(fittingDisaggGenInitRequests.empty()); + EXPECT_TRUE(pausedRequests.empty()); + }; + + // Whisper-like feature encoders have no token IDs to key cross-KV reuse. + // Exercise encoder admission, first decoder context, and chunked decoder + // context, which use the scheduler's three cross-prefix analysis paths. + expectRequestScheduled(); + req->setState(LlmRequestState::kCONTEXT_INIT); + expectRequestScheduled(); + req->setContextCurrentPosition(1); + expectRequestScheduled(); +} From 203b75fed5203864139271fadb5e15e7bf6d6898 Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Mon, 21 Sep 2026 12:00:51 -0700 Subject: [PATCH 3/9] [https://nvbugs/6713231][test] Clarify Whisper reuse coverage Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py b/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py index 2dc76d548648..02f68d4c84ac 100644 --- a/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py +++ b/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py @@ -250,7 +250,7 @@ def test_whisper_pytorch_transcribe_end_to_end(monkeypatch): def test_whisper_pytorch_block_reuse_enabled(monkeypatch): - """Greedy transcription with KV block reuse left at its default (enabled). + """Greedy transcription with KV block reuse explicitly enabled. Whisper requests carry encoder features, not encoder token ids, so the cross-KV pool has nothing to key reuse on; the executor must still admit From 7cb1ad898d481e21121e1c845549b583a381f619 Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Tue, 22 Sep 2026 10:27:30 -0700 Subject: [PATCH 4/9] [https://nvbugs/6713231][fix] Disable unsafe Whisper KV reuse Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- tensorrt_llm/_torch/pyexecutor/_util.py | 16 +++++++++------- .../defs/llmapi/test_llm_api_pytorch_whisper.py | 8 ++++---- tests/integration/test_lists/test-db/l0_h100.yml | 2 +- tests/integration/test_lists/test-db/l0_l40s.yml | 2 +- .../executor/kv_cache/test_dual_pool_kv_cache.py | 12 ++++++++++++ 5 files changed, 27 insertions(+), 13 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/_util.py b/tensorrt_llm/_torch/pyexecutor/_util.py index 7e417a292a09..aba1557aa9d6 100644 --- a/tensorrt_llm/_torch/pyexecutor/_util.py +++ b/tensorrt_llm/_torch/pyexecutor/_util.py @@ -2095,15 +2095,17 @@ def _split_kv_cache_budget_for_cross( self_kv_cache_config = base_kv_cache_config.model_copy() cross_kv_cache_config = base_kv_cache_config.model_copy() - if (cross_kv_cache_config.enable_block_reuse + if (base_kv_cache_config.enable_block_reuse and self._encoder_input_is_features()): - # Cross blocks are keyed on encoder token ids; a feature-driven - # encoder has none, so reuse can never hit and the C++ scheduler - # must not be asked for a cross prefix-reuse summary. + # Decoder self-KV is conditioned on the encoder output, while + # cross-KV is keyed on encoder token ids. Feature-driven encoders + # provide neither reusable token ids nor an input discriminator, + # so neither pool can safely reuse blocks between requests. logger.info( - "Disabling block reuse for the cross-KV cache: the encoder " - "takes feature tensors, so requests carry no encoder tokens " - "to key cross blocks on.") + "Disabling block reuse for the self- and cross-KV caches: " + "the encoder takes feature tensors, so requests carry no " + "encoder tokens or input discriminator to key cache entries.") + self_kv_cache_config.enable_block_reuse = False cross_kv_cache_config.enable_block_reuse = False split_any_budget = False diff --git a/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py b/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py index 02f68d4c84ac..e244e2df9c99 100644 --- a/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py +++ b/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py @@ -249,12 +249,12 @@ def test_whisper_pytorch_transcribe_end_to_end(monkeypatch): ] -def test_whisper_pytorch_block_reuse_enabled(monkeypatch): - """Greedy transcription with KV block reuse explicitly enabled. +def test_whisper_pytorch_block_reuse_requested(monkeypatch): + """Greedy transcription when KV block reuse is requested. Whisper requests carry encoder features, not encoder token ids, so the - cross-KV pool has nothing to key reuse on; the executor must still admit - and run them (https://nvbugs/6713231). Batch 2 co-schedules two + executor disables reuse for both KV pools and must still admit and run + them (https://nvbugs/6713231). Batch 2 co-schedules two encoder-init requests, the shape that reached the unguarded cross-reuse lookup in the C++ capacity scheduler. """ diff --git a/tests/integration/test_lists/test-db/l0_h100.yml b/tests/integration/test_lists/test-db/l0_h100.yml index 1c5b99ff8072..bf417807f2cf 100644 --- a/tests/integration/test_lists/test-db/l0_h100.yml +++ b/tests/integration/test_lists/test-db/l0_h100.yml @@ -193,7 +193,7 @@ l0_h100: - llmapi/test_llm_api_pytorch_t5.py::test_t5_pytorch_generate_encoder_decoder_end_to_end[bf16-kv-v1-cuda-graph-on-beam2-t5-small] - llmapi/test_llm_api_pytorch_t5.py::test_t5_pytorch_generate_encoder_decoder_end_to_end[bf16-kv-v1-cuda-graph-on-greedy-overlap-t5-small] - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_transcribe_end_to_end - - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_block_reuse_enabled + - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_block_reuse_requested - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_feature_combinations[bf16-kv-v2-decoder-graphs-on-greedy] - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_beam_search[bf16-kv-v1-decoder-graphs-on-beam2] - test_e2e.py::test_trtllm_bench_iteration_log[PyTorch-streaming-meta-llama/Llama-3.1-8B-llama-3.1-model/Meta-Llama-3.1-8B] diff --git a/tests/integration/test_lists/test-db/l0_l40s.yml b/tests/integration/test_lists/test-db/l0_l40s.yml index aa70fa119e8a..6bd8a8f588d2 100644 --- a/tests/integration/test_lists/test-db/l0_l40s.yml +++ b/tests/integration/test_lists/test-db/l0_l40s.yml @@ -51,7 +51,7 @@ l0_l40s: # The encoder-graphs case also exercises decoder graphs, so it stands in for # a decoder-only case rather than adding to it; KV-v2 stays covered on H100. - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_transcribe_end_to_end - - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_block_reuse_enabled + - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_block_reuse_requested - llmapi/test_llm_api_pytorch_whisper.py::test_whisper_pytorch_feature_combinations[bf16-kv-v1-encoder-graphs-on-greedy] - condition: ranges: diff --git a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py index a6efac87bba1..9a483a957f3a 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py +++ b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py @@ -289,6 +289,18 @@ def test_split_free_fraction_when_budget_is_zero(self): assert self_config.free_gpu_memory_fraction == pytest.approx(0.4) assert config.free_gpu_memory_fraction == pytest.approx(0.8) + def test_feature_encoder_disables_reuse_for_both_pools(self): + """Feature inputs cannot safely key self- or cross-KV reuse.""" + config = _make_kv_cache_config(cross_kv_cache_fraction=0.5) + creator = _make_creator(config, is_enc_dec=True) + creator._encoder_input_is_features = Mock(return_value=True) + + self_config, cross_config = creator._split_kv_cache_budget_for_cross() + + assert config.enable_block_reuse + assert not self_config.enable_block_reuse + assert not cross_config.enable_block_reuse + def test_is_encoder_decoder_helper(self): dec_config = _make_model_config(is_encoder_decoder=False) dec_creator = _make_creator(_make_kv_cache_config(), model_config=dec_config) From 38e2461df0f7643626ea5bdb900bc8c74d47ac43 Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Wed, 23 Sep 2026 11:13:17 -0700 Subject: [PATCH 5/9] [https://nvbugs/6713231][test] Cover feature encoder manager configs Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- .../kv_cache/test_dual_pool_kv_cache.py | 25 +++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py index 25591e54be4a..ad87588c895b 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py +++ b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py @@ -696,6 +696,31 @@ def create_cross_manager(cross_cfg, *_args, **_kwargs): (pytest.approx(0.45), expected_split), ] + @pytest.mark.parametrize("use_kv_cache_manager_v2", [False, True]) + def test_build_managers_disables_feature_encoder_reuse( + self, use_kv_cache_manager_v2: bool + ) -> None: + """Pass reuse-disabled configs to both feature-encoder managers.""" + config = _make_kv_cache_config( + cross_kv_cache_fraction=0.5, + max_gpu_total_bytes=8 * (1 << 30), + use_kv_cache_manager_v2=use_kv_cache_manager_v2, + ) + creator = _make_creator(config, is_enc_dec=True) + creator.configure_kv_cache_capacity = Mock() + creator._encoder_input_is_features = Mock(return_value=True) + creator._should_create_separate_draft_kv_cache = Mock(return_value=False) + creator._create_kv_cache_manager = Mock(return_value=Mock()) + creator._create_cross_kv_cache_manager = Mock(return_value=Mock()) + + creator.build_managers({}, estimating_kv_cache=False) + + self_config = creator._create_kv_cache_manager.call_args.kwargs["kv_cache_config_override"] + cross_config = creator._create_cross_kv_cache_manager.call_args.args[0] + assert config.enable_block_reuse + assert not self_config.enable_block_reuse + assert not cross_config.enable_block_reuse + def test_build_managers_skips_cross_pool_for_decoder_only(self): creator = _make_creator( _make_kv_cache_config( From 1ede81044c31883512b7c5673161879f0cb2896d Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Wed, 23 Sep 2026 11:25:17 -0700 Subject: [PATCH 6/9] [https://nvbugs/6713231][test] Preserve token encoder reuse coverage Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- .../executor/kv_cache/test_dual_pool_kv_cache.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py index ad87588c895b..570b8ed847f5 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py +++ b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py @@ -302,6 +302,17 @@ def test_feature_encoder_disables_reuse_for_both_pools(self): assert not self_config.enable_block_reuse assert not cross_config.enable_block_reuse + def test_token_encoder_preserves_reuse_for_both_pools(self) -> None: + """Token inputs retain reusable identities for both KV pools.""" + config = _make_kv_cache_config(cross_kv_cache_fraction=0.5) + creator = _make_creator(config, is_enc_dec=True) + creator._encoder_input_is_features = Mock(return_value=False) + + self_config, cross_config = creator._split_kv_cache_budget_for_cross() + + assert self_config.enable_block_reuse + assert cross_config.enable_block_reuse + def test_is_encoder_decoder_helper(self): dec_config = _make_model_config(is_encoder_decoder=False) dec_creator = _make_creator(_make_kv_cache_config(), model_config=dec_config) From ae3ae95622f2b6a7e58e0687ce39da3102b08313 Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:47:06 -0700 Subject: [PATCH 7/9] [https://nvbugs/6713231][fix] Disable paged context for feature encoders Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- tensorrt_llm/_torch/pyexecutor/_util.py | 3 +++ .../_torch/executor/kv_cache/test_dual_pool_kv_cache.py | 6 ++++++ 2 files changed, 9 insertions(+) diff --git a/tensorrt_llm/_torch/pyexecutor/_util.py b/tensorrt_llm/_torch/pyexecutor/_util.py index 2fb318fd349a..7570b8bf840e 100644 --- a/tensorrt_llm/_torch/pyexecutor/_util.py +++ b/tensorrt_llm/_torch/pyexecutor/_util.py @@ -2372,6 +2372,9 @@ def _split_kv_cache_budget_for_cross( "encoder tokens or input discriminator to key cache entries.") self_kv_cache_config.enable_block_reuse = False cross_kv_cache_config.enable_block_reuse = False + # The attention backend reads this shared runtime flag rather than + # either derived manager config when selecting paged-context FMHA. + self._model_engine.attn_runtime_features.cache_reuse = False split_any_budget = False free_fraction = base_kv_cache_config.free_gpu_memory_fraction diff --git a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py index 570b8ed847f5..0903070dca2e 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py +++ b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py @@ -161,6 +161,9 @@ def _make_creator( if model_config is None: model_config = _make_model_config(is_encoder_decoder=is_enc_dec) model_engine = _make_mock_model_engine(model_config) + model_engine.attn_runtime_features = SimpleNamespace( + cache_reuse=kv_cache_config.enable_block_reuse + ) if manager_cls is None: manager_cls = ( @@ -301,6 +304,7 @@ def test_feature_encoder_disables_reuse_for_both_pools(self): assert config.enable_block_reuse assert not self_config.enable_block_reuse assert not cross_config.enable_block_reuse + assert not creator._model_engine.attn_runtime_features.cache_reuse def test_token_encoder_preserves_reuse_for_both_pools(self) -> None: """Token inputs retain reusable identities for both KV pools.""" @@ -312,6 +316,7 @@ def test_token_encoder_preserves_reuse_for_both_pools(self) -> None: assert self_config.enable_block_reuse assert cross_config.enable_block_reuse + assert creator._model_engine.attn_runtime_features.cache_reuse def test_is_encoder_decoder_helper(self): dec_config = _make_model_config(is_encoder_decoder=False) @@ -731,6 +736,7 @@ def test_build_managers_disables_feature_encoder_reuse( assert config.enable_block_reuse assert not self_config.enable_block_reuse assert not cross_config.enable_block_reuse + assert not creator._model_engine.attn_runtime_features.cache_reuse def test_build_managers_skips_cross_pool_for_decoder_only(self): creator = _make_creator( From e6b325788f2e2bdb38839363909ec990b594fabd Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Wed, 23 Sep 2026 14:04:22 -0700 Subject: [PATCH 8/9] [https://nvbugs/6713231][test] Cover Whisper registry and C++ scheduler Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- .../defs/llmapi/test_llm_api_pytorch_whisper.py | 9 +++++++-- .../unittest/_torch/modeling/test_modeling_whisper.py | 10 +++++++++- 2 files changed, 16 insertions(+), 3 deletions(-) diff --git a/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py b/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py index e244e2df9c99..f3860c8461e0 100644 --- a/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py +++ b/tests/integration/defs/llmapi/test_llm_api_pytorch_whisper.py @@ -117,6 +117,7 @@ def _make_llm( tensor_parallel_size: int = 1, encoder_graphs: bool = False, enable_block_reuse: bool = False, + use_python_scheduler: bool = True, ) -> LLM: """Build a Whisper LLM for the test matrix, optionally with encoder CUDA graphs.""" # CudaGraphConfig captures the decode step; the enc-dec encoder step opts in @@ -166,7 +167,7 @@ def _make_llm( # 1500 encoder positions every Whisper request produces. max_input_len=_ENCODER_OUTPUT_LEN, max_num_tokens=2 * _ENCODER_OUTPUT_LEN, - scheduler_config=SchedulerConfig(use_python_scheduler=True), + scheduler_config=SchedulerConfig(use_python_scheduler=use_python_scheduler), tensor_parallel_size=tensor_parallel_size, **encoder_kwargs, **dtype_kwargs, @@ -264,7 +265,11 @@ def test_whisper_pytorch_block_reuse_requested(monkeypatch): wave, sample_rate = soundfile.read(_get_audio_path()) sampling_params = SamplingParams(temperature=0.0, max_tokens=_MAX_NEW_TOKENS) - with _make_llm(model_path, enable_block_reuse=True) as llm: + with _make_llm( + model_path, + enable_block_reuse=True, + use_python_scheduler=False, + ) as llm: for batch_size in (1, 2): outputs = llm.generate( [_audio_prompt(wave, sample_rate) for _ in range(batch_size)], diff --git a/tests/unittest/_torch/modeling/test_modeling_whisper.py b/tests/unittest/_torch/modeling/test_modeling_whisper.py index feb993d258a8..96abd9baf336 100644 --- a/tests/unittest/_torch/modeling/test_modeling_whisper.py +++ b/tests/unittest/_torch/modeling/test_modeling_whisper.py @@ -25,7 +25,15 @@ import torch from transformers import WhisperConfig, WhisperFeatureExtractor -from tensorrt_llm._torch.models.modeling_whisper import WhisperLogMelFrontend +from tensorrt_llm._torch.models.modeling_whisper import ( + WhisperForConditionalGeneration, + WhisperLogMelFrontend, +) +from tensorrt_llm.inputs.registry import input_processor_requires_encoder_features + + +def test_whisper_input_processor_requires_encoder_features(): + assert input_processor_requires_encoder_features(WhisperForConditionalGeneration) def _synthetic_waveform_batch(n_samples: int, seed: int = 1234) -> np.ndarray: From 4d337807276ee49a139361a9ba85adc48b58f8b4 Mon Sep 17 00:00:00 2001 From: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> Date: Wed, 23 Sep 2026 14:16:50 -0700 Subject: [PATCH 9/9] [https://nvbugs/6713231][fix] Detect feature encoders after compilation Signed-off-by: Pranav Shrestha <254760092+pranav-nvidia@users.noreply.github.com> --- tensorrt_llm/_torch/pyexecutor/_util.py | 7 +++---- tensorrt_llm/inputs/registry.py | 12 ------------ .../kv_cache/test_dual_pool_kv_cache.py | 19 ++++++++++++++----- .../_torch/modeling/test_modeling_whisper.py | 8 ++------ 4 files changed, 19 insertions(+), 27 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/_util.py b/tensorrt_llm/_torch/pyexecutor/_util.py index 7570b8bf840e..c48adab79919 100644 --- a/tensorrt_llm/_torch/pyexecutor/_util.py +++ b/tensorrt_llm/_torch/pyexecutor/_util.py @@ -25,8 +25,6 @@ is_sm_100f, prefer_pinned, str_dtype_to_binding, torch_dtype_to_str) from tensorrt_llm.inputs.multimodal import MultimodalParams -from tensorrt_llm.inputs.registry import \ - input_processor_requires_encoder_features # isort: off from tensorrt_llm.llmapi.llm_args import ( @@ -2230,8 +2228,9 @@ def _is_encoder_decoder(self) -> bool: return self._model_engine.model.model_config.is_encoder_decoder def _encoder_input_is_features(self) -> bool: - return input_processor_requires_encoder_features( - type(self._model_engine.model)) + return bool( + getattr(self._model_engine.input_processor, + "requires_encoder_features", False)) @staticmethod def _get_config_int_attr(config, names: tuple[str, ...]) -> Optional[int]: diff --git a/tensorrt_llm/inputs/registry.py b/tensorrt_llm/inputs/registry.py index f63a0f62a640..985075db25ff 100644 --- a/tensorrt_llm/inputs/registry.py +++ b/tensorrt_llm/inputs/registry.py @@ -1074,18 +1074,6 @@ def support_multimodal_disaggregated(model_cls: Type[nn.Module]): return model_cls -def input_processor_requires_encoder_features( - model_cls: Type[nn.Module]) -> bool: - """Whether ``model_cls``'s input processor feeds the encoder a feature tensor. - - Feature-driven encoders (e.g. Whisper's audio encoder) take no encoder - token ids, so such requests have nothing to key cross-KV blocks on. - """ - processor_cls = INPUT_PROCESSOR_REGISTRY._input_processors_cls_by_model_type.get( - model_cls) - return bool(getattr(processor_cls, "requires_encoder_features", False)) - - def register_input_processor( processor_cls: Type[InputProcessor], model_type: str, diff --git a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py index 0903070dca2e..9e1fe4b878e4 100644 --- a/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py +++ b/tests/unittest/_torch/executor/kv_cache/test_dual_pool_kv_cache.py @@ -50,6 +50,14 @@ class _FakeCudaStream: cuda_stream = 0 +class _FakeCompiledModel: + """Minimal torch.compile-style wrapper that changes the model's type.""" + + def __init__(self, original_model): + self._orig_mod = original_model + self.model_config = original_model.model_config + + class _FakeKVCacheManagerCpp: def __init__(self, **kwargs): self.kwargs = kwargs @@ -141,6 +149,7 @@ def _make_mock_model_engine(model_config): engine.dtype = torch.bfloat16 engine.is_draft_model = False engine.kv_cache_manager_key = ResourceManagerType.KV_CACHE_MANAGER + engine.input_processor = SimpleNamespace(requires_encoder_features=False) return engine @@ -293,11 +302,12 @@ def test_split_free_fraction_when_budget_is_zero(self): assert self_config.free_gpu_memory_fraction == pytest.approx(0.4) assert config.free_gpu_memory_fraction == pytest.approx(0.8) - def test_feature_encoder_disables_reuse_for_both_pools(self): - """Feature inputs cannot safely key self- or cross-KV reuse.""" + def test_wrapped_feature_encoder_disables_reuse_for_both_pools(self): + """Feature detection survives a torch.compile-style model wrapper.""" config = _make_kv_cache_config(cross_kv_cache_fraction=0.5) creator = _make_creator(config, is_enc_dec=True) - creator._encoder_input_is_features = Mock(return_value=True) + creator._model_engine.model = _FakeCompiledModel(creator._model_engine.model) + creator._model_engine.input_processor.requires_encoder_features = True self_config, cross_config = creator._split_kv_cache_budget_for_cross() @@ -310,7 +320,6 @@ def test_token_encoder_preserves_reuse_for_both_pools(self) -> None: """Token inputs retain reusable identities for both KV pools.""" config = _make_kv_cache_config(cross_kv_cache_fraction=0.5) creator = _make_creator(config, is_enc_dec=True) - creator._encoder_input_is_features = Mock(return_value=False) self_config, cross_config = creator._split_kv_cache_budget_for_cross() @@ -724,7 +733,7 @@ def test_build_managers_disables_feature_encoder_reuse( ) creator = _make_creator(config, is_enc_dec=True) creator.configure_kv_cache_capacity = Mock() - creator._encoder_input_is_features = Mock(return_value=True) + creator._model_engine.input_processor.requires_encoder_features = True creator._should_create_separate_draft_kv_cache = Mock(return_value=False) creator._create_kv_cache_manager = Mock(return_value=Mock()) creator._create_cross_kv_cache_manager = Mock(return_value=Mock()) diff --git a/tests/unittest/_torch/modeling/test_modeling_whisper.py b/tests/unittest/_torch/modeling/test_modeling_whisper.py index 96abd9baf336..90b13a7daf7c 100644 --- a/tests/unittest/_torch/modeling/test_modeling_whisper.py +++ b/tests/unittest/_torch/modeling/test_modeling_whisper.py @@ -25,15 +25,11 @@ import torch from transformers import WhisperConfig, WhisperFeatureExtractor -from tensorrt_llm._torch.models.modeling_whisper import ( - WhisperForConditionalGeneration, - WhisperLogMelFrontend, -) -from tensorrt_llm.inputs.registry import input_processor_requires_encoder_features +from tensorrt_llm._torch.models.modeling_whisper import WhisperInputProcessor, WhisperLogMelFrontend def test_whisper_input_processor_requires_encoder_features(): - assert input_processor_requires_encoder_features(WhisperForConditionalGeneration) + assert WhisperInputProcessor.requires_encoder_features def _synthetic_waveform_batch(n_samples: int, seed: int = 1234) -> np.ndarray: