fix(router): treat malformed configured token limits as absent on /v1/models (#33864)

A deployment whose model_info carried a non-numeric max_input_tokens or
max_output_tokens (for example "128,000" or an empty string) made the
bare int() in get_configured_token_limits raise inside the per-model
/v1/models loop, so one misconfigured deployment turned the entire
listing into a 500. Coerce each configured limit safely and treat
malformed values as absent, matching the graceful degradation the
listing had before the cost-map switch
This commit is contained in:
yuneng-jiang 2026-07-18 15:27:07 -07:00 • committed by GitHub
parent 9dfd79b6c5
commit ef7007c3dd
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 68 additions and 5 deletions

View file

@ -8535,7 +8535,8 @@ class Router:
Return (max_input_tokens, max_output_tokens) explicitly configured in a concrete
deployment's model_info for model_name, via O(1) index lookup.
Returns (None, None) for wildcard-expanded or unknown names. Unlike
Returns (None, None) for wildcard-expanded or unknown names, and treats a
malformed configured value as absent rather than failing the listing. Unlike
get_model_group_info, this never triggers pattern matching or deep copies, so it
is safe to call per listed model on the /v1/models hot path.
"""
@ -8543,12 +8544,18 @@ class Router:
if deployment is None:
return (None, None)
def _as_int(value: object) -> "int | None":
if value is None or isinstance(value, bool):
return None
try:
return int(value)
except (TypeError, ValueError):
return None
model_info = deployment.model_info
max_input = model_info.get("max_input_tokens")
max_output = model_info.get("max_output_tokens")
return (
int(max_input) if max_input is not None else None,
int(max_output) if max_output is not None else None,
_as_int(model_info.get("max_input_tokens")),
_as_int(model_info.get("max_output_tokens")),
)
def get_deployment_credentials_with_provider(

View file

@ -556,6 +556,31 @@ def test_create_model_info_response_deployment_limits_override_cost_map():
assert response["max_output_tokens"] == 16384
def test_create_model_info_response_survives_malformed_configured_limits():
from litellm import Router
router = Router(
model_list=[
{
"model_name": "bad-limit-model",
"litellm_params": {"model": "openai/some-unmapped-model"},
"model_info": {"max_input_tokens": "128,000"},
}
]
)
response = create_model_info_response(
model_id="bad-limit-model",
provider="openai",
llm_router=router,
get_model_info=_raise_unmapped,
)
assert response["id"] == "bad-limit-model"
assert "max_input_tokens" not in response
assert "max_output_tokens" not in response
def test_create_model_info_response_emits_integer_token_counts():
response = create_model_info_response(
model_id="some-model",

View file

@ -5775,3 +5775,34 @@ def test_get_configured_token_limits_skips_wildcard_pattern_matching():
assert router.get_configured_token_limits(
"bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0"
) == (None, None)
def test_get_configured_token_limits_treats_malformed_values_as_absent():
malformed = ["", "unlimited", "128,000", [128000], {"max": 128000}, True]
router = litellm.Router(
model_list=[
{
"model_name": f"bad-limit-{i}",
"litellm_params": {"model": "openai/some-unmapped-model"},
"model_info": {"max_input_tokens": bad, "max_output_tokens": bad},
}
for i, bad in enumerate(malformed)
]
)
for i in range(len(malformed)):
assert router.get_configured_token_limits(f"bad-limit-{i}") == (None, None)
def test_get_configured_token_limits_coerces_numeric_strings():
router = litellm.Router(
model_list=[
{
"model_name": "quoted-limits-model",
"litellm_params": {"model": "openai/some-unmapped-model"},
"model_info": {"max_input_tokens": "32000", "max_output_tokens": "8000"},
}
]
)
assert router.get_configured_token_limits("quoted-limits-model") == (32000, 8000)