mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(router): treat malformed configured token limits as absent on /v1/models (#33864)
A deployment whose model_info carried a non-numeric max_input_tokens or max_output_tokens (for example "128,000" or an empty string) made the bare int() in get_configured_token_limits raise inside the per-model /v1/models loop, so one misconfigured deployment turned the entire listing into a 500. Coerce each configured limit safely and treat malformed values as absent, matching the graceful degradation the listing had before the cost-map switch
This commit is contained in:
parent
9dfd79b6c5
commit
ef7007c3dd
3 changed files with 68 additions and 5 deletions
|
|
@ -8535,7 +8535,8 @@ class Router:
|
|||
Return (max_input_tokens, max_output_tokens) explicitly configured in a concrete
|
||||
deployment's model_info for model_name, via O(1) index lookup.
|
||||
|
||||
Returns (None, None) for wildcard-expanded or unknown names. Unlike
|
||||
Returns (None, None) for wildcard-expanded or unknown names, and treats a
|
||||
malformed configured value as absent rather than failing the listing. Unlike
|
||||
get_model_group_info, this never triggers pattern matching or deep copies, so it
|
||||
is safe to call per listed model on the /v1/models hot path.
|
||||
"""
|
||||
|
|
@ -8543,12 +8544,18 @@ class Router:
|
|||
if deployment is None:
|
||||
return (None, None)
|
||||
|
||||
def _as_int(value: object) -> "int | None":
|
||||
if value is None or isinstance(value, bool):
|
||||
return None
|
||||
try:
|
||||
return int(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
model_info = deployment.model_info
|
||||
max_input = model_info.get("max_input_tokens")
|
||||
max_output = model_info.get("max_output_tokens")
|
||||
return (
|
||||
int(max_input) if max_input is not None else None,
|
||||
int(max_output) if max_output is not None else None,
|
||||
_as_int(model_info.get("max_input_tokens")),
|
||||
_as_int(model_info.get("max_output_tokens")),
|
||||
)
|
||||
|
||||
def get_deployment_credentials_with_provider(
|
||||
|
|
|
|||
|
|
@ -556,6 +556,31 @@ def test_create_model_info_response_deployment_limits_override_cost_map():
|
|||
assert response["max_output_tokens"] == 16384
|
||||
|
||||
|
||||
def test_create_model_info_response_survives_malformed_configured_limits():
|
||||
from litellm import Router
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "bad-limit-model",
|
||||
"litellm_params": {"model": "openai/some-unmapped-model"},
|
||||
"model_info": {"max_input_tokens": "128,000"},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response = create_model_info_response(
|
||||
model_id="bad-limit-model",
|
||||
provider="openai",
|
||||
llm_router=router,
|
||||
get_model_info=_raise_unmapped,
|
||||
)
|
||||
|
||||
assert response["id"] == "bad-limit-model"
|
||||
assert "max_input_tokens" not in response
|
||||
assert "max_output_tokens" not in response
|
||||
|
||||
|
||||
def test_create_model_info_response_emits_integer_token_counts():
|
||||
response = create_model_info_response(
|
||||
model_id="some-model",
|
||||
|
|
|
|||
|
|
@ -5775,3 +5775,34 @@ def test_get_configured_token_limits_skips_wildcard_pattern_matching():
|
|||
assert router.get_configured_token_limits(
|
||||
"bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0"
|
||||
) == (None, None)
|
||||
|
||||
|
||||
def test_get_configured_token_limits_treats_malformed_values_as_absent():
|
||||
malformed = ["", "unlimited", "128,000", [128000], {"max": 128000}, True]
|
||||
router = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": f"bad-limit-{i}",
|
||||
"litellm_params": {"model": "openai/some-unmapped-model"},
|
||||
"model_info": {"max_input_tokens": bad, "max_output_tokens": bad},
|
||||
}
|
||||
for i, bad in enumerate(malformed)
|
||||
]
|
||||
)
|
||||
|
||||
for i in range(len(malformed)):
|
||||
assert router.get_configured_token_limits(f"bad-limit-{i}") == (None, None)
|
||||
|
||||
|
||||
def test_get_configured_token_limits_coerces_numeric_strings():
|
||||
router = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "quoted-limits-model",
|
||||
"litellm_params": {"model": "openai/some-unmapped-model"},
|
||||
"model_info": {"max_input_tokens": "32000", "max_output_tokens": "8000"},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
assert router.get_configured_token_limits("quoted-limits-model") == (32000, 8000)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue