mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
docs: state the mixed-group precedence for strict_token_count
A deployment carrying strict_token_count: false does not short-circuit to permissive; it falls through to the router check, so a group is strict if any deployment asks for it. Greptile flagged this as non-obvious. Documented on the helper and pinned with a test, so it is a contract rather than an accident.
This commit is contained in:
parent
00d656c341
commit
cf98a33374
2 changed files with 59 additions and 0 deletions
|
|
@ -11704,6 +11704,14 @@ def _is_strict_token_count_model(
|
|||
router's configuration for the requested model. The fallback matters
|
||||
because deployment selection can fail for reasons unrelated to the
|
||||
policy, and a strict model must not quietly return an estimate then.
|
||||
|
||||
Note the precedence when a model group is configured inconsistently: a
|
||||
deployment carrying `strict_token_count: false` does **not** short-circuit
|
||||
to permissive. It falls through to the router check, so the group is
|
||||
strict if any of its deployments asks for it. The caller cannot choose
|
||||
which deployment serves them, so an ambiguous configuration is read the
|
||||
safe way. Set the flag on every deployment in a group to avoid relying on
|
||||
this.
|
||||
"""
|
||||
if model_info is not None and bool(model_info.get("strict_token_count", False)):
|
||||
return True
|
||||
|
|
|
|||
|
|
@ -235,6 +235,57 @@ async def test_non_strict_model_still_estimates_when_selection_fails():
|
|||
assert response.total_tokens > 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_mixed_group_is_strict_if_any_deployment_asks():
|
||||
"""A group configured inconsistently resolves to strict.
|
||||
|
||||
A deployment carrying `strict_token_count: false` does not make the group
|
||||
permissive when a sibling asks for strict. The caller cannot choose which
|
||||
deployment serves them, so the ambiguous configuration is read the safe
|
||||
way. Documented on `_is_strict_token_count_model` because it is
|
||||
non-obvious. See https://github.com/BerriAI/litellm/issues/37102.
|
||||
"""
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "mixed-model",
|
||||
"litellm_params": {
|
||||
"model": "bedrock/anthropic.claude-opus-5-20260101-v1:0",
|
||||
"aws_region_name": "us-east-1",
|
||||
"aws_access_key_id": "fake",
|
||||
"aws_secret_access_key": "fake",
|
||||
},
|
||||
"model_info": {"strict_token_count": False},
|
||||
},
|
||||
{
|
||||
"model_name": "mixed-model",
|
||||
"litellm_params": {
|
||||
"model": "bedrock/anthropic.claude-opus-5-20260101-v1:0",
|
||||
"aws_region_name": "us-west-2",
|
||||
"aws_access_key_id": "fake",
|
||||
"aws_secret_access_key": "fake",
|
||||
},
|
||||
"model_info": {"strict_token_count": True},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
original_router = getattr(proxy_server, "llm_router", None)
|
||||
setattr(proxy_server, "llm_router", router)
|
||||
try:
|
||||
with _unsupported_count_tokens():
|
||||
with pytest.raises(ProxyException):
|
||||
await token_counter(
|
||||
request=TokenCountRequest(
|
||||
model="mixed-model",
|
||||
messages=[{"role": "user", "content": "hello " * 400}],
|
||||
),
|
||||
call_endpoint=True,
|
||||
)
|
||||
finally:
|
||||
setattr(proxy_server, "llm_router", original_router)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_disable_token_counter_still_applies_proxy_wide():
|
||||
"""The existing proxy-wide flag must keep working for unmarked models."""
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue