From ab02127b50a26ce9cd2a2ff33eed6a0ab7ff3123 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Sat, 18 Jul 2026 18:56:24 -0700 Subject: [PATCH] fix(proxy): treat malformed cost-map token limits as absent on /v1/models create_model_info_response cast cost-map max_input_tokens / max_output_tokens with unguarded int(). The surrounding try/except covers only the get_model_info lookup, so a deployment whose model_info carries a non-numeric limit (e.g. "128,000" or an empty string) raised inside the per-model listing loop and failed the entire GET /v1/models and /models response with a 500, taking healthy deployments down with it. A deployment's model_info is registered into litellm.model_cost verbatim, so the malformed value reaches the cost map and not just the router index. Router.get_configured_token_limits already coerced this safely for the deployment path; the cost-map path was missed, so the two together still regressed. Both now share coerce_token_limit in litellm_core_utils, which returns None for a malformed value so the listing omits that one limit instead of failing, matching the graceful degradation the endpoint had before the cost-map switch. --- litellm/litellm_core_utils/core_helpers.py | 29 +++++++++ litellm/proxy/utils.py | 9 +-- litellm/router.py | 13 +--- tests/test_litellm/proxy/test_proxy_utils.py | 64 +++++++++++++++++++- 4 files changed, 98 insertions(+), 17 deletions(-) diff --git a/litellm/litellm_core_utils/core_helpers.py b/litellm/litellm_core_utils/core_helpers.py index 002a46771e3..88dddb59cc7 100644 --- a/litellm/litellm_core_utils/core_helpers.py +++ b/litellm/litellm_core_utils/core_helpers.py @@ -57,6 +57,35 @@ def safe_divide( return numerator / denominator +def coerce_token_limit(value: object) -> int | None: + """ + Coerce a max_input_tokens / max_output_tokens value to an int, treating a + malformed value as absent. + + A deployment's model_info is registered into litellm.model_cost verbatim, so a + config value like "128,000" or "" reaches the /v1/models listing uncoerced from + both the router index and the cost map. Returning None omits that one limit + instead of failing the whole listing. + + Args: + value: The raw configured or cost-map value + + Returns: + The value as an int, or None if it is missing or not a usable number. + Bools are rejected because True/False is never a meaningful token limit. + """ + if isinstance(value, bool): + return None + if isinstance(value, int): + return value + if isinstance(value, (str, float)): + try: + return int(value) + except (TypeError, ValueError, OverflowError): + return None + return None + + _FINISH_REASON_MAP: dict[str, OpenAIChatCompletionFinishReason] = { # Anthropic "stop_sequence": "stop", diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 7a52fdfdb87..844d3c2ed26 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -104,6 +104,7 @@ from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.prometheus import PrometheusLogger from litellm.integrations.SlackAlerting.slack_alerting import SlackAlerting from litellm.integrations.SlackAlerting.utils import _add_langfuse_trace_id_to_alert +from litellm.litellm_core_utils.core_helpers import coerce_token_limit from litellm.litellm_core_utils.litellm_logging import Logging from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.litellm_core_utils.safe_json_loads import safe_json_loads @@ -6128,12 +6129,8 @@ def create_model_info_response( max_input_tokens: int | None = None max_output_tokens: int | None = None if model_cost_info is not None: - cost_map_input = model_cost_info.get("max_input_tokens") - if cost_map_input is not None: - max_input_tokens = int(cost_map_input) - cost_map_output = model_cost_info.get("max_output_tokens") - if cost_map_output is not None: - max_output_tokens = int(cost_map_output) + max_input_tokens = coerce_token_limit(model_cost_info.get("max_input_tokens")) + max_output_tokens = coerce_token_limit(model_cost_info.get("max_output_tokens")) if llm_router is not None: configured_input, configured_output = llm_router.get_configured_token_limits(model_id) diff --git a/litellm/router.py b/litellm/router.py index 9e44edb1fb9..ae3f7ba11c2 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -68,6 +68,7 @@ from litellm.litellm_core_utils.request_timeout_resolver import ( ) from litellm.litellm_core_utils.core_helpers import ( _get_parent_otel_span_from_kwargs, + coerce_token_limit, get_metadata_variable_name_from_kwargs, ) from litellm.litellm_core_utils.coroutine_checker import coroutine_checker @@ -8544,18 +8545,10 @@ class Router: if deployment is None: return (None, None) - def _as_int(value: object) -> "int | None": - if value is None or isinstance(value, bool): - return None - try: - return int(value) - except (TypeError, ValueError): - return None - model_info = deployment.model_info return ( - _as_int(model_info.get("max_input_tokens")), - _as_int(model_info.get("max_output_tokens")), + coerce_token_limit(model_info.get("max_input_tokens")), + coerce_token_limit(model_info.get("max_output_tokens")), ) def get_deployment_credentials_with_provider( diff --git a/tests/test_litellm/proxy/test_proxy_utils.py b/tests/test_litellm/proxy/test_proxy_utils.py index 9486646ea4a..5ace46fc775 100644 --- a/tests/test_litellm/proxy/test_proxy_utils.py +++ b/tests/test_litellm/proxy/test_proxy_utils.py @@ -478,11 +478,12 @@ class TestPostCallFailureHookLiftsRecoveredPartialSpend: from typing import cast +import litellm from litellm.proxy.utils import create_model_info_response from litellm.types.utils import ModelInfo -def _fake_model_info(**fields: int) -> ModelInfo: +def _fake_model_info(**fields: object) -> ModelInfo: return cast(ModelInfo, dict(fields)) @@ -581,6 +582,67 @@ def test_create_model_info_response_survives_malformed_configured_limits(): assert "max_output_tokens" not in response +@pytest.mark.parametrize("bad_value", ["128,000", "", "unlimited", [128000], {"max": 128000}, True]) +def test_create_model_info_response_survives_malformed_cost_map_limits(bad_value): + response = create_model_info_response( + model_id="some-model", + provider="openai", + llm_router=None, + get_model_info=lambda _model: _fake_model_info( + max_input_tokens=bad_value, max_output_tokens=bad_value + ), + ) + + assert response["id"] == "some-model" + assert "max_input_tokens" not in response + assert "max_output_tokens" not in response + + +def test_create_model_info_response_keeps_valid_cost_map_limit_beside_malformed_one(): + response = create_model_info_response( + model_id="some-model", + provider="openai", + llm_router=None, + get_model_info=lambda _model: _fake_model_info( + max_input_tokens="128,000", max_output_tokens=16384 + ), + ) + + assert "max_input_tokens" not in response + assert response["max_output_tokens"] == 16384 + + +def test_create_model_info_response_survives_malformed_limits_registered_by_router(): + """A deployment's model_info is registered into litellm.model_cost verbatim, so a + malformed configured limit reaches the listing through the real cost-map lookup and + not just the router index. Guarding only the index path still 500s the whole listing.""" + from litellm import Router + + saved_model_cost = dict(litellm.model_cost) + try: + router = Router( + model_list=[ + { + "model_name": "openai/some-unmapped-model", + "litellm_params": {"model": "openai/some-unmapped-model"}, + "model_info": {"max_input_tokens": "128,000"}, + } + ] + ) + + response = create_model_info_response( + model_id="openai/some-unmapped-model", + provider="openai", + llm_router=router, + ) + finally: + litellm.model_cost.clear() + litellm.model_cost.update(saved_model_cost) + + assert response["id"] == "openai/some-unmapped-model" + assert "max_input_tokens" not in response + + def test_create_model_info_response_emits_integer_token_counts(): response = create_model_info_response( model_id="some-model",