From fc5848174e48c56280a0f2892a0c8a5dc4b03ed8 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Thu, 16 Jul 2026 19:53:10 -0700 Subject: [PATCH] fix(router): take the lowest minimum across a model group, not the highest The read gate cannot cause a wrong pin. A deployment is only pinned when the cache already holds an entry for the prefix, and async_log_success_event writes entries against the deployment's real model rather than the group alias, so a model that will not cache a prefix never records one and there is nothing to pin it to That makes this gate purely a cheap short-circuit deciding whether the cache lookup is worth doing, so the threshold must be the lowest minimum in the group. Taking the highest skipped the lookup for a prefix a lower-minimum member had genuinely cached, losing a hit it earned, and protected against nothing. It also broke the Fable 5 direction this ticket is meant to fix: its real minimum is 512, so a group gate stuck at a higher value would skip the lookup for a prefix Fable 5 had actually cached --- .../prompt_caching_deployment_check.py | 18 +++++++----- .../test_prompt_caching_deployment_check.py | 29 +++++++++++++++---- 2 files changed, 34 insertions(+), 13 deletions(-) diff --git a/litellm/router_utils/pre_call_checks/prompt_caching_deployment_check.py b/litellm/router_utils/pre_call_checks/prompt_caching_deployment_check.py index 1d121d79ea3..d6412c95da0 100644 --- a/litellm/router_utils/pre_call_checks/prompt_caching_deployment_check.py +++ b/litellm/router_utils/pre_call_checks/prompt_caching_deployment_check.py @@ -19,16 +19,20 @@ from ..prompt_caching_cache import PromptCachingCache def _get_min_token_count_for_deployments(healthy_deployments: list[dict]) -> int: """ - Returns the highest minimum cacheable prefix across a model group. + Returns the lowest minimum cacheable prefix across a model group. + This gate only decides whether the cache lookup is worth doing. It cannot cause a wrong pin, + because a deployment is only pinned when the cache already holds an entry for the prefix, and + entries are written by `async_log_success_event` against the deployment's real model. A model + that will not cache a prefix never records one, so there is nothing to pin it to. + + That makes the lowest minimum in the group the correct threshold rather than the highest. `model` here is the model-group alias the operator chose, not a model name, so the threshold - has to come from the deployments themselves. A group may mix models with different minimums, - and one gate decides for all of them, so take the max: a prompt is only treated as cacheable - when it clears every member's minimum. The errors are not symmetric. Pinning a deployment for - a prefix its provider will not cache costs load balancing for nothing, which is the bug this - guards against, while declining to pin only forfeits a cache hit. + has to come from the deployments themselves, and a group may mix models whose minimums differ. + Taking the highest would skip the lookup for a prefix a lower-minimum member genuinely cached, + losing a cache hit it had earned. The lowest can only cost a lookup that finds nothing. """ - return max( + return min( ( get_prompt_cache_min_tokens(model=deployment["litellm_params"]["model"]) for deployment in healthy_deployments diff --git a/tests/test_litellm/router_utils/pre_call_checks/test_prompt_caching_deployment_check.py b/tests/test_litellm/router_utils/pre_call_checks/test_prompt_caching_deployment_check.py index 6ad928b9737..6752d76847f 100644 --- a/tests/test_litellm/router_utils/pre_call_checks/test_prompt_caching_deployment_check.py +++ b/tests/test_litellm/router_utils/pre_call_checks/test_prompt_caching_deployment_check.py @@ -15,7 +15,7 @@ from litellm.router_utils.pre_call_checks.prompt_caching_deployment_check import ) from litellm.router_utils.prompt_caching_cache import PromptCachingCache from litellm.types.llms.openai import AllMessageValues -from litellm.utils import get_prompt_cache_min_tokens, token_counter +from litellm.utils import get_prompt_cache_min_tokens, is_prompt_caching_valid_prompt, token_counter MODEL_GROUP_ALIAS = "my-claude-group" OPUS_4_6_MIN_TOKENS = 4096 @@ -72,18 +72,35 @@ def _messages(word_count: int) -> List[AllMessageValues]: ) -def test_get_min_token_count_for_deployments_takes_max_across_mixed_group(): +def test_get_min_token_count_for_deployments_takes_min_across_mixed_group(): """ - A group may legally mix models whose real minimums differ, and one boolean gate decides for - every member. The threshold must be the highest minimum in the group: taking the lowest would - let a 1024-token prompt pin the Opus 4.5 deployment for a prefix Anthropic will never cache. + A group may legally mix models whose real minimums differ, and one gate decides for every + member. The threshold must be the lowest minimum in the group. This gate only decides whether + the cache lookup happens, so taking the highest would skip the lookup for a prefix the Sonnet + 4.5 deployment genuinely cached and lose a hit it had earned. """ assert get_prompt_cache_min_tokens(model="anthropic/claude-opus-4-5") == 4096 assert get_prompt_cache_min_tokens(model="anthropic/claude-sonnet-4-5") == 1024 deployments = _deployments("anthropic/claude-opus-4-5", "anthropic/claude-sonnet-4-5") - assert _get_min_token_count_for_deployments(deployments) == 4096 + assert _get_min_token_count_for_deployments(deployments) == 1024 + + +def test_write_gate_is_what_prevents_a_pin_below_the_model_minimum(): + """ + The invariant the read gate relies on. A deployment can only be pinned when the cache already + holds an entry for the prefix, and `async_log_success_event` writes entries against the real + deployment model. Opus 4.5 never records an entry for a prefix it will not cache, so no read + threshold is what keeps it from being pinned. + """ + messages = _messages(word_count=1400) + + token_count = token_counter(messages=messages, model="anthropic/claude-opus-4-5", use_default_image_token_count=True) + assert 1024 < token_count < 4096 + + assert is_prompt_caching_valid_prompt(model="anthropic/claude-opus-4-5", messages=messages) is False + assert is_prompt_caching_valid_prompt(model="anthropic/claude-sonnet-4-5", messages=messages) is True def test_get_min_token_count_for_deployments_falls_back_to_default_for_empty_group():