From cf98a333748736f2480f0fa76ff7a8706136f17d Mon Sep 17 00:00:00 2001 From: aayushbaluni <73417844+aayushbaluni@users.noreply.github.com> Date: Tue, 18 Aug 2026 19:55:58 +0530 Subject: [PATCH] docs: state the mixed-group precedence for strict_token_count A deployment carrying strict_token_count: false does not short-circuit to permissive; it falls through to the router check, so a group is strict if any deployment asks for it. Greptile flagged this as non-obvious. Documented on the helper and pinned with a test, so it is a contract rather than an accident. --- litellm/proxy/proxy_server.py | 8 +++ .../test_strict_token_count.py | 51 +++++++++++++++++++ 2 files changed, 59 insertions(+) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 88f99664be8..b74c0872ce2 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -11704,6 +11704,14 @@ def _is_strict_token_count_model( router's configuration for the requested model. The fallback matters because deployment selection can fail for reasons unrelated to the policy, and a strict model must not quietly return an estimate then. + + Note the precedence when a model group is configured inconsistently: a + deployment carrying `strict_token_count: false` does **not** short-circuit + to permissive. It falls through to the router check, so the group is + strict if any of its deployments asks for it. The caller cannot choose + which deployment serves them, so an ambiguous configuration is read the + safe way. Set the flag on every deployment in a group to avoid relying on + this. """ if model_info is not None and bool(model_info.get("strict_token_count", False)): return True diff --git a/tests/proxy_unit_tests/test_strict_token_count.py b/tests/proxy_unit_tests/test_strict_token_count.py index 66cfc3d67ae..6d30f907582 100644 --- a/tests/proxy_unit_tests/test_strict_token_count.py +++ b/tests/proxy_unit_tests/test_strict_token_count.py @@ -235,6 +235,57 @@ async def test_non_strict_model_still_estimates_when_selection_fails(): assert response.total_tokens > 0 +@pytest.mark.asyncio +async def test_mixed_group_is_strict_if_any_deployment_asks(): + """A group configured inconsistently resolves to strict. + + A deployment carrying `strict_token_count: false` does not make the group + permissive when a sibling asks for strict. The caller cannot choose which + deployment serves them, so the ambiguous configuration is read the safe + way. Documented on `_is_strict_token_count_model` because it is + non-obvious. See https://github.com/BerriAI/litellm/issues/37102. + """ + router = Router( + model_list=[ + { + "model_name": "mixed-model", + "litellm_params": { + "model": "bedrock/anthropic.claude-opus-5-20260101-v1:0", + "aws_region_name": "us-east-1", + "aws_access_key_id": "fake", + "aws_secret_access_key": "fake", + }, + "model_info": {"strict_token_count": False}, + }, + { + "model_name": "mixed-model", + "litellm_params": { + "model": "bedrock/anthropic.claude-opus-5-20260101-v1:0", + "aws_region_name": "us-west-2", + "aws_access_key_id": "fake", + "aws_secret_access_key": "fake", + }, + "model_info": {"strict_token_count": True}, + }, + ] + ) + + original_router = getattr(proxy_server, "llm_router", None) + setattr(proxy_server, "llm_router", router) + try: + with _unsupported_count_tokens(): + with pytest.raises(ProxyException): + await token_counter( + request=TokenCountRequest( + model="mixed-model", + messages=[{"role": "user", "content": "hello " * 400}], + ), + call_endpoint=True, + ) + finally: + setattr(proxy_server, "llm_router", original_router) + + @pytest.mark.asyncio async def test_disable_token_counter_still_applies_proxy_wide(): """The existing proxy-wide flag must keep working for unmarked models."""