From f6ec7dfeb9e1ee7026d22d98c3dc4851b89c754e Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 15:27:18 +0200 Subject: [PATCH 01/10] feat(router): treat_finish_reason_as_failure config surface and helpers --- litellm/router.py | 112 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 112 insertions(+) diff --git a/litellm/router.py b/litellm/router.py index 633f060f208..a2aec6847a3 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -752,6 +752,7 @@ class Router: fallbacks: list = [], context_window_fallbacks: list = [], content_policy_fallbacks: list = [], + treat_finish_reason_as_failure: dict[str, str] | None = None, model_group_alias: dict[str, str | RouterModelGroupAliasItem] | None = {}, enable_pre_call_checks: bool = False, enable_tag_filtering: bool = False, @@ -1072,6 +1073,30 @@ class Router: _content_policy_fallbacks: Final = content_policy_fallbacks or litellm.content_policy_fallbacks self.validate_fallbacks(fallback_param=_content_policy_fallbacks) self.content_policy_fallbacks = _content_policy_fallbacks + + ## treat_finish_reason_as_failure: map a terminal finish/stop reason on a 200 response to a + ## router-understood exception class, so the mapped reason engages allowed_fails/cooldowns/ + ## fallbacks like any failure. Values must name one of: RateLimitError, APIError, + ## BadRequestError, Timeout, ServiceUnavailableError, InternalServerError (resolved from + ## litellm at use time). Reason strings are matched exactly. + _finish_reason_failure_exception_names: Final = frozenset( + { + "RateLimitError", + "APIError", + "BadRequestError", + "Timeout", + "ServiceUnavailableError", + "InternalServerError", + } + ) + if treat_finish_reason_as_failure is not None: + for exception_name in treat_finish_reason_as_failure.values(): + if exception_name not in _finish_reason_failure_exception_names: + raise ValueError( + f"treat_finish_reason_as_failure values must be one of {sorted(_finish_reason_failure_exception_names)}, got {exception_name}" + ) + self.treat_finish_reason_as_failure = treat_finish_reason_as_failure + self.total_calls: defaultdict = defaultdict(int) # dict to store total calls made to each model self.fail_calls: defaultdict = defaultdict(int) # dict to store fail_calls made to each model self.success_calls: defaultdict = defaultdict(int) # dict to store success_calls made to each model @@ -8591,6 +8616,93 @@ class Router: ) return resolved is not None + def _get_mapped_finish_reason(self, response: ModelResponse) -> str | None: + """ + The finish reason configured in treat_finish_reason_as_failure that this response carries, + or None. Checks both the mapped finish_reason and the pre-mapping value stashed in + provider_specific_fields["native_finish_reason"]. Streaming detection is a follow-up + modeled on _aanthropic_messages_streaming_iterator. + """ + if not self.treat_finish_reason_as_failure: + return None + if not (response.choices and len(response.choices) > 0): + return None + choice: Final = response.choices[0] + if choice.finish_reason in self.treat_finish_reason_as_failure: + return choice.finish_reason + native_reason: Final = (choice.provider_specific_fields or {}).get("native_finish_reason") + if native_reason in self.treat_finish_reason_as_failure: + return native_reason + return None + + def _finish_reason_failure_fallback_available(self, model_group: str, kwargs: Mapping[str, Any]) -> bool: + """ + Whether a generic fallback can serve the retry after a mapped finish-reason failure. + Mirrors the tail of _refusal_fallback_available without the content-policy branch: the + dispatcher falls through to the generic fallbacks lookup, so the gate arms on default + fallbacks or a resolving generic chain. + """ + if fallbacks_disabled_for_request(kwargs): + return False + if self._has_default_fallbacks(): + return True + fallbacks: Final = kwargs.get("fallbacks", self.fallbacks) + if fallbacks is None: + return False + resolved, _ = get_fallback_model_group_for_lookup_groups( + fallbacks=fallbacks, + lookup_groups=fallback_lookup_groups(kwargs, model_group), + ) + return resolved is not None + + def _should_raise_mapped_finish_reason_error(self, model: str, response: ModelResponse, kwargs: dict) -> bool: + """ + True when the response carries a reason from treat_finish_reason_as_failure and a generic + fallback can serve the retry. When a reason is mapped but no fallback is available the + caller must still account for the failure via _account_mapped_finish_reason_failure. + """ + if self._get_mapped_finish_reason(response) is None: + return False + return self._finish_reason_failure_fallback_available(model, kwargs) + + def _finish_reason_failure_error(self, model: str, reason: str) -> Exception: + """Build the exception instance configured for a mapped finish reason.""" + exception_name: Final = self.treat_finish_reason_as_failure[reason] + exception_cls: Final = getattr(litellm, exception_name) + message: Final = f"Response finished with reason '{reason}' (treat_finish_reason_as_failure)." + if exception_name == "APIError": + return exception_cls(status_code=500, message=message, llm_provider="", model=model) + return exception_cls(message=message, llm_provider="", model=model) + + def _account_mapped_finish_reason_failure( + self, model: str, deployment: dict, response: ModelResponse, kwargs: dict + ) -> None: + """ + Count and park a mapped finish-reason failure when nothing is raised (no fallback can + serve the retry): increment the per-minute failure counter and set the cooldown the way a + raised exception would. In the raising branch the normal exception flow does the accounting. + """ + reason: Final = self._get_mapped_finish_reason(response) + if reason is None: + return + model_info: Final = deployment.get("model_info") or {} + deployment_id: Final = model_info.get("id") + if deployment_id is None: + return + exception: Final = self._finish_reason_failure_error(model=model, reason=reason) + increment_deployment_failures_for_current_minute( + litellm_router_instance=self, + deployment_id=deployment_id, + ) + _set_cooldown_deployments( + litellm_router_instance=self, + exception_status=exception.status_code, + original_exception=exception, + deployment=deployment_id, + time_to_cooldown=self.cooldown_time, + requested_model_group=(get_litellm_metadata_from_kwargs(kwargs) or {}).get("model_group"), + ) + def _should_raise_content_policy_error(self, model: str, response: ModelResponse, kwargs: dict) -> bool: """ Determines if a content policy error should be raised. From ebe4967783d0037fed4f242e39443138081e827d Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 15:28:13 +0200 Subject: [PATCH 02/10] feat(router): mapped finish reason checkpoint on chat completions --- litellm/router.py | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/litellm/router.py b/litellm/router.py index a2aec6847a3..aa36ddf7eef 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -2574,6 +2574,16 @@ class Router: llm_provider="", ) + ## CHECK MAPPED FINISH REASON ERROR ## + if isinstance(response, ModelResponse): + if self._should_raise_mapped_finish_reason_error(model=model, response=response, kwargs=kwargs): + raise self._finish_reason_failure_error( + model=model, reason=self._get_mapped_finish_reason(response) + ) + self._account_mapped_finish_reason_failure( + model=model, deployment=deployment, response=response, kwargs=kwargs + ) + if ( isinstance(response, CustomStreamWrapper) and response.completion_stream is None @@ -3698,6 +3708,16 @@ class Router: llm_provider="", ) + ## CHECK MAPPED FINISH REASON ERROR ## + if isinstance(response, ModelResponse): + if self._should_raise_mapped_finish_reason_error(model=model, response=response, kwargs=kwargs): + raise self._finish_reason_failure_error( + model=model, reason=self._get_mapped_finish_reason(response) + ) + self._account_mapped_finish_reason_failure( + model=model, deployment=deployment, response=response, kwargs=kwargs + ) + if ( isinstance(response, CustomStreamWrapper) and response.completion_stream is None From 3d15b1d412989204a3dd71adbe9c2a467ce00190 Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 15:30:10 +0200 Subject: [PATCH 03/10] feat(router): mapped finish reason check on anthropic_messages dispatch --- litellm/router.py | 26 ++++++++++++++++++++------ 1 file changed, 20 insertions(+), 6 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index aa36ddf7eef..f73bfc6b7da 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -2581,7 +2581,10 @@ class Router: model=model, reason=self._get_mapped_finish_reason(response) ) self._account_mapped_finish_reason_failure( - model=model, deployment=deployment, response=response, kwargs=kwargs + model=model, + deployment=deployment, + reason=self._get_mapped_finish_reason(response), + kwargs=kwargs, ) if ( @@ -3715,7 +3718,10 @@ class Router: model=model, reason=self._get_mapped_finish_reason(response) ) self._account_mapped_finish_reason_failure( - model=model, deployment=deployment, response=response, kwargs=kwargs + model=model, + deployment=deployment, + reason=self._get_mapped_finish_reason(response), + kwargs=kwargs, ) if ( @@ -5435,6 +5441,17 @@ class Router: refusal_details: Final = cast(dict, response["stop_details"]) # cast-ok: gate verified the shape raise safeguard_refusal_error(model=model, stop_details=refusal_details) + if getattr(original_generic_function, "__name__", "") == "anthropic_messages" and isinstance( + response, dict + ): + stop_reason: Final = response.get("stop_reason") + if stop_reason in (self.treat_finish_reason_as_failure or {}): + if self._finish_reason_failure_fallback_available(model, kwargs): + raise self._finish_reason_failure_error(model=model, reason=stop_reason) + self._account_mapped_finish_reason_failure( + model=model, deployment=deployment, reason=stop_reason, kwargs=kwargs + ) + self.success_calls[model_name] += 1 verbose_router_logger.info("ageneric_api_call_with_fallbacks(model=%s)\x1b[32m 200 OK\x1b[0m", model_name) @@ -8694,15 +8711,12 @@ class Router: return exception_cls(status_code=500, message=message, llm_provider="", model=model) return exception_cls(message=message, llm_provider="", model=model) - def _account_mapped_finish_reason_failure( - self, model: str, deployment: dict, response: ModelResponse, kwargs: dict - ) -> None: + def _account_mapped_finish_reason_failure(self, model: str, deployment: dict, reason: str, kwargs: dict) -> None: """ Count and park a mapped finish-reason failure when nothing is raised (no fallback can serve the retry): increment the per-minute failure counter and set the cooldown the way a raised exception would. In the raising branch the normal exception flow does the accounting. """ - reason: Final = self._get_mapped_finish_reason(response) if reason is None: return model_info: Final = deployment.get("model_info") or {} From 773870eed7a2908a7e8797defa93b99ba7a75148 Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 15:48:31 +0200 Subject: [PATCH 04/10] feat(router): stash native finish reason on anthropic transform and account on raise --- litellm/llms/anthropic/chat/transformation.py | 13 +++++- litellm/router.py | 44 +++++++++---------- 2 files changed, 32 insertions(+), 25 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 0f99441a115..11e502a073f 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -2626,10 +2626,21 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): model_response.choices[0].message = _message model_response._hidden_params["original_response"] = completion_response["content"] - model_response.choices[0].finish_reason = cast( + _finish_reason: Final = cast( OpenAIChatCompletionFinishReason, map_finish_reason(completion_response["stop_reason"]), ) + model_response.choices[0].finish_reason = _finish_reason + if completion_response["stop_reason"] and completion_response["stop_reason"] != _finish_reason: + _choice = model_response.choices[0] + setattr( + _choice, + "provider_specific_fields", + { + **(getattr(_choice, "provider_specific_fields", None) or {}), + "native_finish_reason": completion_response["stop_reason"], + }, + ) usage: Final = self.calculate_usage( usage_object=completion_response["usage"], diff --git a/litellm/router.py b/litellm/router.py index f73bfc6b7da..c12ed3ea397 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -2576,16 +2576,13 @@ class Router: ## CHECK MAPPED FINISH REASON ERROR ## if isinstance(response, ModelResponse): - if self._should_raise_mapped_finish_reason_error(model=model, response=response, kwargs=kwargs): - raise self._finish_reason_failure_error( - model=model, reason=self._get_mapped_finish_reason(response) + _mapped_reason = self._get_mapped_finish_reason(response) + if _mapped_reason is not None: + self._account_mapped_finish_reason_failure( + model=model, deployment=deployment, reason=_mapped_reason, kwargs=kwargs ) - self._account_mapped_finish_reason_failure( - model=model, - deployment=deployment, - reason=self._get_mapped_finish_reason(response), - kwargs=kwargs, - ) + if self._should_raise_mapped_finish_reason_error(model=model, response=response, kwargs=kwargs): + raise self._finish_reason_failure_error(model=model, reason=_mapped_reason) if ( isinstance(response, CustomStreamWrapper) @@ -3713,16 +3710,15 @@ class Router: ## CHECK MAPPED FINISH REASON ERROR ## if isinstance(response, ModelResponse): - if self._should_raise_mapped_finish_reason_error(model=model, response=response, kwargs=kwargs): - raise self._finish_reason_failure_error( - model=model, reason=self._get_mapped_finish_reason(response) + _mapped_reason = self._get_mapped_finish_reason(response) + if _mapped_reason is not None: + self._account_mapped_finish_reason_failure( + model=model, deployment=deployment, reason=_mapped_reason, kwargs=kwargs ) - self._account_mapped_finish_reason_failure( - model=model, - deployment=deployment, - reason=self._get_mapped_finish_reason(response), - kwargs=kwargs, - ) + if self._should_raise_mapped_finish_reason_error( + model=model, response=response, kwargs=kwargs + ): + raise self._finish_reason_failure_error(model=model, reason=_mapped_reason) if ( isinstance(response, CustomStreamWrapper) @@ -5446,11 +5442,11 @@ class Router: ): stop_reason: Final = response.get("stop_reason") if stop_reason in (self.treat_finish_reason_as_failure or {}): - if self._finish_reason_failure_fallback_available(model, kwargs): - raise self._finish_reason_failure_error(model=model, reason=stop_reason) self._account_mapped_finish_reason_failure( model=model, deployment=deployment, reason=stop_reason, kwargs=kwargs ) + if self._finish_reason_failure_fallback_available(model, kwargs): + raise self._finish_reason_failure_error(model=model, reason=stop_reason) self.success_calls[model_name] += 1 verbose_router_logger.info("ageneric_api_call_with_fallbacks(model=%s)\x1b[32m 200 OK\x1b[0m", model_name) @@ -8667,7 +8663,7 @@ class Router: choice: Final = response.choices[0] if choice.finish_reason in self.treat_finish_reason_as_failure: return choice.finish_reason - native_reason: Final = (choice.provider_specific_fields or {}).get("native_finish_reason") + native_reason: Final = (getattr(choice, "provider_specific_fields", None) or {}).get("native_finish_reason") if native_reason in self.treat_finish_reason_as_failure: return native_reason return None @@ -8713,9 +8709,9 @@ class Router: def _account_mapped_finish_reason_failure(self, model: str, deployment: dict, reason: str, kwargs: dict) -> None: """ - Count and park a mapped finish-reason failure when nothing is raised (no fallback can - serve the retry): increment the per-minute failure counter and set the cooldown the way a - raised exception would. In the raising branch the normal exception flow does the accounting. + Count and park a mapped finish-reason failure: increment the per-minute failure counter + and set the cooldown. The raise sites call this too, because the router raises after the + 200 came back, so litellm's failure callbacks never fire for this exception. """ if reason is None: return From 3a67e2f44f2c299eef1c58e92951273731710b3c Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 15:48:32 +0200 Subject: [PATCH 05/10] test(router): finish reason failure coverage --- .../test_router_finish_reason_failure.py | 180 ++++++++++++++++++ 1 file changed, 180 insertions(+) create mode 100644 tests/router_unit_tests/test_router_finish_reason_failure.py diff --git a/tests/router_unit_tests/test_router_finish_reason_failure.py b/tests/router_unit_tests/test_router_finish_reason_failure.py new file mode 100644 index 00000000000..a11fafa5557 --- /dev/null +++ b/tests/router_unit_tests/test_router_finish_reason_failure.py @@ -0,0 +1,180 @@ +""" +Unit tests for the treat_finish_reason_as_failure router knob. + +A provider can report a terminal condition (context window exceeded, and +similar) as a stop reason on an HTTP 200. The knob maps such reasons to a +router-understood exception class, so the mapped reason engages allowed_fails, +cooldowns, and fallbacks like any failure. When the mapped reason is present +but no generic fallback can serve the retry, the response reaches the client +unchanged while the deployment still counts the failure. +""" + +import json +from typing import Any + +import httpx +import pytest +from pytest import MonkeyPatch + +from litellm import Router +from litellm.router_utils.cooldown_handlers import _get_cooldown_deployments + +CONTEXT_WINDOW_RESPONSE: dict[str, Any] = { + "id": "msg_context", + "type": "message", + "role": "assistant", + "model": "claude-fable-5", + "content": [], + "stop_reason": "model_context_window_exceeded", + "stop_sequence": None, + "usage": {"input_tokens": 25, "output_tokens": 1}, +} + +OK_RESPONSE: dict[str, Any] = { + "id": "msg_ok", + "type": "message", + "role": "assistant", + "model": "claude-opus-5", + "content": [{"type": "text", "text": "hello"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 25, "output_tokens": 2}, +} + + +class FakeAnthropicUpstream: + """Intercepts the third-party transport (httpx.AsyncClient.send): reports the + context-window stop reason on fable models, answers on others. The router + deliberately does not forward caller-injected clients, so the transport is the + seam that exercises the real litellm pipeline end to end.""" + + def __init__(self) -> None: + self.calls: list[str] = [] + + async def send(self, request: httpx.Request, **kwargs: Any) -> httpx.Response: + body = json.loads(request.content or b"{}") + model = body.get("model", "") + self.calls.append(model) + overrun = "fable" in model + return httpx.Response( + 200, + json=CONTEXT_WINDOW_RESPONSE if overrun else OK_RESPONSE, + request=request, + ) + + def install(self, monkeypatch: MonkeyPatch) -> None: + async def _send(_client: httpx.AsyncClient, request: httpx.Request, **kwargs: Any) -> httpx.Response: + return await self.send(request, **kwargs) + + monkeypatch.setattr(httpx.AsyncClient, "send", _send) + + +FABLE_TIER = { + "model_name": "fable-tier", + "litellm_params": {"model": "anthropic/claude-fable-5", "api_key": "sk-test"}, +} +OPUS_TARGET = { + "model_name": "opus-target", + "litellm_params": {"model": "anthropic/claude-opus-5", "api_key": "sk-test"}, +} + + +def _knob() -> dict[str, str]: + return {"model_context_window_exceeded": "RateLimitError"} + + +def _deployment_id(router: Router, index: int = 0) -> str: + return router.model_list[index]["model_info"]["id"] + + +@pytest.mark.asyncio +async def test_chat_completion_mapped_reason_falls_back_and_cools_down(monkeypatch: MonkeyPatch): + fake = FakeAnthropicUpstream() + router = Router( + model_list=[FABLE_TIER, OPUS_TARGET], + treat_finish_reason_as_failure=_knob(), + default_fallbacks=["opus-target"], + num_retries=0, + allowed_fails=0, + cooldown_time=10, + ) + fake.install(monkeypatch) + + response = await router.acompletion( + model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}] + ) + + assert response.model == "claude-opus-5" + assert len(fake.calls) == 2 + assert "claude-fable-5" in fake.calls[0] + assert "claude-opus-5" in fake.calls[1] + assert router.fail_calls["anthropic/claude-fable-5"] == 1 + fable_id = _deployment_id(router, 0) + assert fable_id in _get_cooldown_deployments(litellm_router_instance=router, parent_otel_span=None) + + +@pytest.mark.asyncio +async def test_anthropic_messages_mapped_reason_falls_back(monkeypatch: MonkeyPatch): + fake = FakeAnthropicUpstream() + router = Router( + model_list=[FABLE_TIER, OPUS_TARGET], + treat_finish_reason_as_failure=_knob(), + default_fallbacks=["opus-target"], + num_retries=0, + allowed_fails=0, + cooldown_time=10, + ) + fake.install(monkeypatch) + + response = await router.aanthropic_messages( + model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}] + ) + + assert response["id"] == "msg_ok" + assert response["stop_reason"] == "end_turn" + assert len(fake.calls) == 2 + assert "claude-opus-5" in fake.calls[1] + + +@pytest.mark.asyncio +async def test_mapped_reason_without_fallback_returns_response_and_counts_failure(monkeypatch: MonkeyPatch): + fake = FakeAnthropicUpstream() + router = Router( + model_list=[FABLE_TIER], + treat_finish_reason_as_failure=_knob(), + num_retries=0, + allowed_fails=0, + cooldown_time=10, + ) + fake.install(monkeypatch) + + response = await router.acompletion( + model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}] + ) + + assert response.model == "claude-fable-5" + assert len(fake.calls) == 1 + fable_id = _deployment_id(router, 0) + failures = router.cache.get_cache(local_only=True, key=f"{fable_id}:fails") + assert failures == 1 + assert fable_id in _get_cooldown_deployments(litellm_router_instance=router, parent_otel_span=None) + + +@pytest.mark.asyncio +async def test_knob_unset_ignores_terminal_stop_reason(monkeypatch: MonkeyPatch): + fake = FakeAnthropicUpstream() + router = Router(model_list=[FABLE_TIER], num_retries=0) + fake.install(monkeypatch) + + response = await router.acompletion( + model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}] + ) + + assert response.model == "claude-fable-5" + assert len(fake.calls) == 1 + assert router.fail_calls["fable-tier"] == 0 + + +def test_unknown_exception_name_raises_at_construction(): + with pytest.raises(ValueError, match="NotAnException"): + Router(model_list=[], treat_finish_reason_as_failure={"x": "NotAnException"}) From 46035650f23496ed3bfb97c1f281f53b4b858450 Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 16:44:18 +0200 Subject: [PATCH 06/10] refactor(router): consolidate mapped finish reason helpers - one _handle_mapped_finish_reason_failure (account, gate, raise) replaces the duplicated checkpoint blocks and the double exception build - _generic_fallback_available shared by the refusal and knob availability gates - accounting honors a deployment-level cooldown_time like deployment_callback_on_failure - map+stash of the native finish reason lives in one helper (map_finish_reason_and_stash_native) shared by Choices.__init__ and the anthropic transform - knob-off requests pay a single None check at every checkpoint --- litellm/llms/anthropic/chat/transformation.py | 22 ++-- litellm/router.py | 119 +++++++++--------- litellm/types/utils.py | 20 ++- 3 files changed, 83 insertions(+), 78 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 11e502a073f..3eb7a33594f 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -22,7 +22,6 @@ from litellm.constants import ( DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, RESPONSE_FORMAT_TOOL_NAME, ) -from litellm.litellm_core_utils.core_helpers import map_finish_reason from litellm.litellm_core_utils.prompt_templates.common_utils import ( sanitize_input_schema_for_anthropic, ) @@ -75,6 +74,7 @@ from litellm.types.responses.main import ( from litellm.types.utils import ( CacheCreationTokenDetails, CompletionTokensDetailsWrapper, + map_finish_reason_and_stash_native, PromptTokensDetailsWrapper, ServerToolUse, ) @@ -2626,21 +2626,13 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): model_response.choices[0].message = _message model_response._hidden_params["original_response"] = completion_response["content"] - _finish_reason: Final = cast( - OpenAIChatCompletionFinishReason, - map_finish_reason(completion_response["stop_reason"]), + _choice = model_response.choices[0] + _mapped_reason, _provider_specific_fields = map_finish_reason_and_stash_native( + completion_response["stop_reason"], getattr(_choice, "provider_specific_fields", None) ) - model_response.choices[0].finish_reason = _finish_reason - if completion_response["stop_reason"] and completion_response["stop_reason"] != _finish_reason: - _choice = model_response.choices[0] - setattr( - _choice, - "provider_specific_fields", - { - **(getattr(_choice, "provider_specific_fields", None) or {}), - "native_finish_reason": completion_response["stop_reason"], - }, - ) + _choice.finish_reason = _mapped_reason + if _provider_specific_fields is not None: + setattr(_choice, "provider_specific_fields", _provider_specific_fields) usage: Final = self.calculate_usage( usage_object=completion_response["usage"], diff --git a/litellm/router.py b/litellm/router.py index c12ed3ea397..5326ad99bc5 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -708,6 +708,20 @@ def as_output_cap(value: object) -> int | None: return cap if cap >= 0 else None +## Exception classes a treat_finish_reason_as_failure value may name: resolved from litellm at +## use time, validated at Router construction. +_FINISH_REASON_FAILURE_EXCEPTION_NAMES: Final = frozenset( + { + "RateLimitError", + "APIError", + "BadRequestError", + "Timeout", + "ServiceUnavailableError", + "InternalServerError", + } +) + + class Router: model_names: set = set() cache_responses: bool | None = False @@ -1076,24 +1090,12 @@ class Router: ## treat_finish_reason_as_failure: map a terminal finish/stop reason on a 200 response to a ## router-understood exception class, so the mapped reason engages allowed_fails/cooldowns/ - ## fallbacks like any failure. Values must name one of: RateLimitError, APIError, - ## BadRequestError, Timeout, ServiceUnavailableError, InternalServerError (resolved from - ## litellm at use time). Reason strings are matched exactly. - _finish_reason_failure_exception_names: Final = frozenset( - { - "RateLimitError", - "APIError", - "BadRequestError", - "Timeout", - "ServiceUnavailableError", - "InternalServerError", - } - ) + ## fallbacks like any failure. Reason strings are matched exactly. if treat_finish_reason_as_failure is not None: for exception_name in treat_finish_reason_as_failure.values(): - if exception_name not in _finish_reason_failure_exception_names: + if exception_name not in _FINISH_REASON_FAILURE_EXCEPTION_NAMES: raise ValueError( - f"treat_finish_reason_as_failure values must be one of {sorted(_finish_reason_failure_exception_names)}, got {exception_name}" + f"treat_finish_reason_as_failure values must be one of {sorted(_FINISH_REASON_FAILURE_EXCEPTION_NAMES)}, got {exception_name}" ) self.treat_finish_reason_as_failure = treat_finish_reason_as_failure @@ -2578,11 +2580,9 @@ class Router: if isinstance(response, ModelResponse): _mapped_reason = self._get_mapped_finish_reason(response) if _mapped_reason is not None: - self._account_mapped_finish_reason_failure( + self._handle_mapped_finish_reason_failure( model=model, deployment=deployment, reason=_mapped_reason, kwargs=kwargs ) - if self._should_raise_mapped_finish_reason_error(model=model, response=response, kwargs=kwargs): - raise self._finish_reason_failure_error(model=model, reason=_mapped_reason) if ( isinstance(response, CustomStreamWrapper) @@ -3712,13 +3712,9 @@ class Router: if isinstance(response, ModelResponse): _mapped_reason = self._get_mapped_finish_reason(response) if _mapped_reason is not None: - self._account_mapped_finish_reason_failure( + self._handle_mapped_finish_reason_failure( model=model, deployment=deployment, reason=_mapped_reason, kwargs=kwargs ) - if self._should_raise_mapped_finish_reason_error( - model=model, response=response, kwargs=kwargs - ): - raise self._finish_reason_failure_error(model=model, reason=_mapped_reason) if ( isinstance(response, CustomStreamWrapper) @@ -5437,16 +5433,16 @@ class Router: refusal_details: Final = cast(dict, response["stop_details"]) # cast-ok: gate verified the shape raise safeguard_refusal_error(model=model, stop_details=refusal_details) - if getattr(original_generic_function, "__name__", "") == "anthropic_messages" and isinstance( - response, dict + if ( + self.treat_finish_reason_as_failure + and getattr(original_generic_function, "__name__", "") == "anthropic_messages" + and isinstance(response, dict) ): stop_reason: Final = response.get("stop_reason") - if stop_reason in (self.treat_finish_reason_as_failure or {}): - self._account_mapped_finish_reason_failure( + if stop_reason in self.treat_finish_reason_as_failure: + self._handle_mapped_finish_reason_failure( model=model, deployment=deployment, reason=stop_reason, kwargs=kwargs ) - if self._finish_reason_failure_fallback_available(model, kwargs): - raise self._finish_reason_failure_error(model=model, reason=stop_reason) self.success_calls[model_name] += 1 verbose_router_logger.info("ageneric_api_call_with_fallbacks(model=%s)\x1b[32m 200 OK\x1b[0m", model_name) @@ -8638,16 +8634,7 @@ class Router: content_policy_fallbacks: Final = kwargs.get("content_policy_fallbacks", self.content_policy_fallbacks) if content_policy_fallbacks is not None: return self._has_content_policy_fallback(model_group, kwargs) - if self._has_default_fallbacks(): - return True - fallbacks: Final = kwargs.get("fallbacks", self.fallbacks) - if fallbacks is None: - return False - resolved, _ = get_fallback_model_group_for_lookup_groups( - fallbacks=fallbacks, - lookup_groups=fallback_lookup_groups(kwargs, model_group), - ) - return resolved is not None + return self._generic_fallback_available(model_group, kwargs) def _get_mapped_finish_reason(self, response: ModelResponse) -> str | None: """ @@ -8668,12 +8655,10 @@ class Router: return native_reason return None - def _finish_reason_failure_fallback_available(self, model_group: str, kwargs: Mapping[str, Any]) -> bool: + def _generic_fallback_available(self, model_group: str, kwargs: Mapping[str, Any]) -> bool: """ - Whether a generic fallback can serve the retry after a mapped finish-reason failure. - Mirrors the tail of _refusal_fallback_available without the content-policy branch: the - dispatcher falls through to the generic fallbacks lookup, so the gate arms on default - fallbacks or a resolving generic chain. + Whether a generic fallback can serve a retry: default fallbacks set, or a generic chain + resolving for this request. Shared tail of the fallback-availability gates. """ if fallbacks_disabled_for_request(kwargs): return False @@ -8688,15 +8673,18 @@ class Router: ) return resolved is not None - def _should_raise_mapped_finish_reason_error(self, model: str, response: ModelResponse, kwargs: dict) -> bool: + def _handle_mapped_finish_reason_failure(self, model: str, deployment: dict, reason: str, kwargs: dict) -> None: """ - True when the response carries a reason from treat_finish_reason_as_failure and a generic - fallback can serve the retry. When a reason is mapped but no fallback is available the - caller must still account for the failure via _account_mapped_finish_reason_failure. + Account for a mapped finish-reason failure, then raise the configured exception into the + fallback chain when a generic fallback can serve. Accounting happens before the gate: + the raise lands after the 200 came back, so litellm's failure callbacks never fire for + it, and this is the only path that parks the deployment. """ - if self._get_mapped_finish_reason(response) is None: - return False - return self._finish_reason_failure_fallback_available(model, kwargs) + exception: Final = self._account_mapped_finish_reason_failure( + model=model, deployment=deployment, reason=reason, kwargs=kwargs + ) + if exception is not None and self._generic_fallback_available(model, kwargs): + raise exception def _finish_reason_failure_error(self, model: str, reason: str) -> Exception: """Build the exception instance configured for a mapped finish reason.""" @@ -8707,18 +8695,30 @@ class Router: return exception_cls(status_code=500, message=message, llm_provider="", model=model) return exception_cls(message=message, llm_provider="", model=model) - def _account_mapped_finish_reason_failure(self, model: str, deployment: dict, reason: str, kwargs: dict) -> None: + def _account_mapped_finish_reason_failure( + self, model: str, deployment: dict, reason: str, kwargs: dict + ) -> Exception | None: """ Count and park a mapped finish-reason failure: increment the per-minute failure counter - and set the cooldown. The raise sites call this too, because the router raises after the - 200 came back, so litellm's failure callbacks never fire for this exception. + and set the cooldown, honoring a deployment-level cooldown_time like + deployment_callback_on_failure does (the retry-after-header tier has no counterpart + here: the exception is synthesized, it carries no response headers). Returns the built + exception so the caller can raise the same instance it accounted for, or None when the + deployment has no id to account against. """ - if reason is None: - return model_info: Final = deployment.get("model_info") or {} - deployment_id: Final = model_info.get("id") + deployment_id: Final = model_info.get("id") if isinstance(model_info, dict) else None if deployment_id is None: - return + return None + litellm_params: Final = deployment.get("litellm_params") or {} + deployment_cooldown: Final = _first_present( + model_info if isinstance(model_info, dict) else None, litellm_params, key="cooldown_time" + ) + time_to_cooldown: Final = ( + deployment_cooldown + if deployment_cooldown is not None and deployment_cooldown >= 0 + else self.cooldown_time + ) exception: Final = self._finish_reason_failure_error(model=model, reason=reason) increment_deployment_failures_for_current_minute( litellm_router_instance=self, @@ -8729,9 +8729,10 @@ class Router: exception_status=exception.status_code, original_exception=exception, deployment=deployment_id, - time_to_cooldown=self.cooldown_time, + time_to_cooldown=time_to_cooldown, requested_model_group=(get_litellm_metadata_from_kwargs(kwargs) or {}).get("model_group"), ) + return exception def _should_raise_content_policy_error(self, model: str, response: ModelResponse, kwargs: dict) -> bool: """ diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 748c91a4792..5bf0a4e38e9 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1567,6 +1567,19 @@ class Delta(SafeAttributeModel, OpenAIObject): setattr(self, key, value) +def map_finish_reason_and_stash_native( + finish_reason: str, provider_specific_fields: dict[str, Any] | None +) -> tuple[OpenAIChatCompletionFinishReason, dict[str, Any] | None]: + """Map a provider-native finish reason to the OpenAI set; when the native value differs + from the mapped one, preserve it under provider_specific_fields["native_finish_reason"] + so downstream consumers can still see what the provider actually sent.""" + mapped: Final = map_finish_reason(finish_reason) + if finish_reason != mapped: + provider_specific_fields = dict(provider_specific_fields) if provider_specific_fields else {} + provider_specific_fields["native_finish_reason"] = finish_reason + return mapped, provider_specific_fields + + class Choices(SafeAttributeModel, OpenAIObject): finish_reason: OpenAIChatCompletionFinishReason index: int @@ -1586,11 +1599,10 @@ class Choices(SafeAttributeModel, OpenAIObject): **params, ) -> None: if finish_reason is not None: - mapped: Final = map_finish_reason(finish_reason) + mapped, provider_specific_fields = map_finish_reason_and_stash_native( + finish_reason, provider_specific_fields + ) params["finish_reason"] = mapped - if finish_reason != mapped: - provider_specific_fields = dict(provider_specific_fields) if provider_specific_fields else {} - provider_specific_fields["native_finish_reason"] = finish_reason else: params["finish_reason"] = "stop" if index is not None: From bde1fb662ce8637c95830e3921635c883478f2d6 Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 17:06:47 +0200 Subject: [PATCH 07/10] fix(router): knob init warnings and raise-without-deployment-id - warn at construction that the knob applies to non-streaming responses only - warn when a key names a healthy terminal reason in the mapped OpenAI set - a deployment without model_info id still raises into the fallback chain when one can serve; only the accounting needs the id --- litellm/router.py | 22 +++++++++++++++++++--- 1 file changed, 19 insertions(+), 3 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 5326ad99bc5..09d7ea622ed 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -1098,6 +1098,21 @@ class Router: f"treat_finish_reason_as_failure values must be one of {sorted(_FINISH_REASON_FAILURE_EXCEPTION_NAMES)}, got {exception_name}" ) self.treat_finish_reason_as_failure = treat_finish_reason_as_failure + if treat_finish_reason_as_failure: + verbose_router_logger.warning( + "treat_finish_reason_as_failure applies to non-streaming responses only; a streamed 200 with the mapped stop reason is delivered unchanged." + ) + healthy_terminal_keys: Final = treat_finish_reason_as_failure.keys() & { + "stop", + "length", + "tool_calls", + "function_call", + } + if healthy_terminal_keys: + verbose_router_logger.warning( + "treat_finish_reason_as_failure keys %s are healthy terminal reasons in the mapped OpenAI set; mapping them fails successful responses. Keys are matched against provider-native stop reasons.", + sorted(healthy_terminal_keys), + ) self.total_calls: defaultdict = defaultdict(int) # dict to store total calls made to each model self.fail_calls: defaultdict = defaultdict(int) # dict to store fail_calls made to each model @@ -8678,12 +8693,13 @@ class Router: Account for a mapped finish-reason failure, then raise the configured exception into the fallback chain when a generic fallback can serve. Accounting happens before the gate: the raise lands after the 200 came back, so litellm's failure callbacks never fire for - it, and this is the only path that parks the deployment. + it, and this is the only path that parks the deployment. A deployment with no model_info + id cannot be accounted or parked, but the raise still applies to it. """ exception: Final = self._account_mapped_finish_reason_failure( model=model, deployment=deployment, reason=reason, kwargs=kwargs - ) - if exception is not None and self._generic_fallback_available(model, kwargs): + ) or self._finish_reason_failure_error(model=model, reason=reason) + if self._generic_fallback_available(model, kwargs): raise exception def _finish_reason_failure_error(self, model: str, reason: str) -> Exception: From f9a1359ca282ffcf2695ee00842fc26a475674cd Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 17:19:38 +0200 Subject: [PATCH 08/10] chore: ruff format and import order --- litellm/llms/anthropic/chat/transformation.py | 3 +-- litellm/router.py | 4 +--- .../test_router_finish_reason_failure.py | 12 +++--------- 3 files changed, 5 insertions(+), 14 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 3eb7a33594f..ef780b372e1 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -62,7 +62,6 @@ from litellm.types.llms.openai import ( ChatCompletionToolCallChunk, ChatCompletionToolCallFunctionChunk, ChatCompletionToolParam, - OpenAIChatCompletionFinishReason, OpenAIMcpServerTool, OpenAIWebSearchOptions, ) @@ -74,9 +73,9 @@ from litellm.types.responses.main import ( from litellm.types.utils import ( CacheCreationTokenDetails, CompletionTokensDetailsWrapper, - map_finish_reason_and_stash_native, PromptTokensDetailsWrapper, ServerToolUse, + map_finish_reason_and_stash_native, ) from litellm.types.utils import Message as LitellmMessage from litellm.utils import ( diff --git a/litellm/router.py b/litellm/router.py index 09d7ea622ed..10c0102df5c 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8731,9 +8731,7 @@ class Router: model_info if isinstance(model_info, dict) else None, litellm_params, key="cooldown_time" ) time_to_cooldown: Final = ( - deployment_cooldown - if deployment_cooldown is not None and deployment_cooldown >= 0 - else self.cooldown_time + deployment_cooldown if deployment_cooldown is not None and deployment_cooldown >= 0 else self.cooldown_time ) exception: Final = self._finish_reason_failure_error(model=model, reason=reason) increment_deployment_failures_for_current_minute( diff --git a/tests/router_unit_tests/test_router_finish_reason_failure.py b/tests/router_unit_tests/test_router_finish_reason_failure.py index a11fafa5557..280e7defcea 100644 --- a/tests/router_unit_tests/test_router_finish_reason_failure.py +++ b/tests/router_unit_tests/test_router_finish_reason_failure.py @@ -100,9 +100,7 @@ async def test_chat_completion_mapped_reason_falls_back_and_cools_down(monkeypat ) fake.install(monkeypatch) - response = await router.acompletion( - model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}] - ) + response = await router.acompletion(model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}]) assert response.model == "claude-opus-5" assert len(fake.calls) == 2 @@ -148,9 +146,7 @@ async def test_mapped_reason_without_fallback_returns_response_and_counts_failur ) fake.install(monkeypatch) - response = await router.acompletion( - model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}] - ) + response = await router.acompletion(model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}]) assert response.model == "claude-fable-5" assert len(fake.calls) == 1 @@ -166,9 +162,7 @@ async def test_knob_unset_ignores_terminal_stop_reason(monkeypatch: MonkeyPatch) router = Router(model_list=[FABLE_TIER], num_retries=0) fake.install(monkeypatch) - response = await router.acompletion( - model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}] - ) + response = await router.acompletion(model="fable-tier", max_tokens=16, messages=[{"role": "user", "content": "hi"}]) assert response.model == "claude-fable-5" assert len(fake.calls) == 1 From 2210f948aba506d97b37960ea61f20b1ac1f8651 Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 17:45:15 +0200 Subject: [PATCH 09/10] fix(router): balance the type-discipline budget and cover the knob helpers by name - Mapping annotations and Final locals keep every LIT rule at or below the base count - the stash helper builds under a new name; its mutable copy carries a reason - direct helper calls in the test satisfy the router code-coverage name check --- litellm/router.py | 44 +++++++++++-------- litellm/types/utils.py | 13 +++--- .../test_router_finish_reason_failure.py | 40 +++++++++++++++++ 3 files changed, 72 insertions(+), 25 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 10c0102df5c..87d3d95d646 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -721,6 +721,10 @@ _FINISH_REASON_FAILURE_EXCEPTION_NAMES: Final = frozenset( } ) +## Healthy terminal reasons in the mapped OpenAI set: keys of treat_finish_reason_as_failure that +## name one of these would fail successful responses, so construction warns about them. +_HEALTHY_TERMINAL_FINISH_REASONS: Final = frozenset(("stop", "length", "tool_calls", "function_call")) + class Router: model_names: set = set() @@ -766,7 +770,7 @@ class Router: fallbacks: list = [], context_window_fallbacks: list = [], content_policy_fallbacks: list = [], - treat_finish_reason_as_failure: dict[str, str] | None = None, + treat_finish_reason_as_failure: Mapping[str, str] | None = None, model_group_alias: dict[str, str | RouterModelGroupAliasItem] | None = {}, enable_pre_call_checks: bool = False, enable_tag_filtering: bool = False, @@ -1102,12 +1106,7 @@ class Router: verbose_router_logger.warning( "treat_finish_reason_as_failure applies to non-streaming responses only; a streamed 200 with the mapped stop reason is delivered unchanged." ) - healthy_terminal_keys: Final = treat_finish_reason_as_failure.keys() & { - "stop", - "length", - "tool_calls", - "function_call", - } + healthy_terminal_keys: Final = treat_finish_reason_as_failure.keys() & _HEALTHY_TERMINAL_FINISH_REASONS if healthy_terminal_keys: verbose_router_logger.warning( "treat_finish_reason_as_failure keys %s are healthy terminal reasons in the mapped OpenAI set; mapping them fails successful responses. Keys are matched against provider-native stop reasons.", @@ -2593,7 +2592,7 @@ class Router: ## CHECK MAPPED FINISH REASON ERROR ## if isinstance(response, ModelResponse): - _mapped_reason = self._get_mapped_finish_reason(response) + _mapped_reason: Final = self._get_mapped_finish_reason(response) if _mapped_reason is not None: self._handle_mapped_finish_reason_failure( model=model, deployment=deployment, reason=_mapped_reason, kwargs=kwargs @@ -3725,7 +3724,7 @@ class Router: ## CHECK MAPPED FINISH REASON ERROR ## if isinstance(response, ModelResponse): - _mapped_reason = self._get_mapped_finish_reason(response) + _mapped_reason: Final = self._get_mapped_finish_reason(response) if _mapped_reason is not None: self._handle_mapped_finish_reason_failure( model=model, deployment=deployment, reason=_mapped_reason, kwargs=kwargs @@ -8665,7 +8664,10 @@ class Router: choice: Final = response.choices[0] if choice.finish_reason in self.treat_finish_reason_as_failure: return choice.finish_reason - native_reason: Final = (getattr(choice, "provider_specific_fields", None) or {}).get("native_finish_reason") + _provider_specific_fields: Final = getattr(choice, "provider_specific_fields", None) + native_reason: Final = ( + _provider_specific_fields.get("native_finish_reason") if _provider_specific_fields else None + ) if native_reason in self.treat_finish_reason_as_failure: return native_reason return None @@ -8688,7 +8690,9 @@ class Router: ) return resolved is not None - def _handle_mapped_finish_reason_failure(self, model: str, deployment: dict, reason: str, kwargs: dict) -> None: + def _handle_mapped_finish_reason_failure( + self, model: str, deployment: Mapping[str, Any], reason: str, kwargs: Mapping[str, Any] + ) -> None: """ Account for a mapped finish-reason failure, then raise the configured exception into the fallback chain when a generic fallback can serve. Accounting happens before the gate: @@ -8712,7 +8716,7 @@ class Router: return exception_cls(message=message, llm_provider="", model=model) def _account_mapped_finish_reason_failure( - self, model: str, deployment: dict, reason: str, kwargs: dict + self, model: str, deployment: Mapping[str, Any], reason: str, kwargs: Mapping[str, Any] ) -> Exception | None: """ Count and park a mapped finish-reason failure: increment the per-minute failure counter @@ -8722,18 +8726,20 @@ class Router: exception so the caller can raise the same instance it accounted for, or None when the deployment has no id to account against. """ - model_info: Final = deployment.get("model_info") or {} - deployment_id: Final = model_info.get("id") if isinstance(model_info, dict) else None + raw_model_info: Final = deployment.get("model_info") + model_info: Final = raw_model_info if isinstance(raw_model_info, dict) else None + deployment_id: Final = model_info.get("id") if model_info is not None else None if deployment_id is None: return None - litellm_params: Final = deployment.get("litellm_params") or {} - deployment_cooldown: Final = _first_present( - model_info if isinstance(model_info, dict) else None, litellm_params, key="cooldown_time" - ) + raw_litellm_params: Final = deployment.get("litellm_params") + litellm_params: Final = raw_litellm_params if isinstance(raw_litellm_params, dict) else None + deployment_cooldown: Final = _first_present(model_info, litellm_params, key="cooldown_time") time_to_cooldown: Final = ( deployment_cooldown if deployment_cooldown is not None and deployment_cooldown >= 0 else self.cooldown_time ) exception: Final = self._finish_reason_failure_error(model=model, reason=reason) + litellm_metadata: Final = get_litellm_metadata_from_kwargs(kwargs) + requested_model_group: Final = litellm_metadata.get("model_group") if litellm_metadata else None increment_deployment_failures_for_current_minute( litellm_router_instance=self, deployment_id=deployment_id, @@ -8744,7 +8750,7 @@ class Router: original_exception=exception, deployment=deployment_id, time_to_cooldown=time_to_cooldown, - requested_model_group=(get_litellm_metadata_from_kwargs(kwargs) or {}).get("model_group"), + requested_model_group=requested_model_group, ) return exception diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 5bf0a4e38e9..9dac60c9095 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1568,16 +1568,17 @@ class Delta(SafeAttributeModel, OpenAIObject): def map_finish_reason_and_stash_native( - finish_reason: str, provider_specific_fields: dict[str, Any] | None -) -> tuple[OpenAIChatCompletionFinishReason, dict[str, Any] | None]: + finish_reason: str, provider_specific_fields: Mapping[str, Any] | None +) -> tuple[OpenAIChatCompletionFinishReason, dict[str, Any] | None]: # mutable-ok: callers extend the returned stash """Map a provider-native finish reason to the OpenAI set; when the native value differs from the mapped one, preserve it under provider_specific_fields["native_finish_reason"] so downstream consumers can still see what the provider actually sent.""" mapped: Final = map_finish_reason(finish_reason) - if finish_reason != mapped: - provider_specific_fields = dict(provider_specific_fields) if provider_specific_fields else {} - provider_specific_fields["native_finish_reason"] = finish_reason - return mapped, provider_specific_fields + if finish_reason == mapped: + return mapped, provider_specific_fields + stash: Final = dict(provider_specific_fields or ()) # mutable-ok: the stash must stay a plain extensible dict + stash["native_finish_reason"] = finish_reason + return mapped, stash class Choices(SafeAttributeModel, OpenAIObject): diff --git a/tests/router_unit_tests/test_router_finish_reason_failure.py b/tests/router_unit_tests/test_router_finish_reason_failure.py index 280e7defcea..5c9fd639313 100644 --- a/tests/router_unit_tests/test_router_finish_reason_failure.py +++ b/tests/router_unit_tests/test_router_finish_reason_failure.py @@ -16,6 +16,7 @@ import httpx import pytest from pytest import MonkeyPatch +import litellm from litellm import Router from litellm.router_utils.cooldown_handlers import _get_cooldown_deployments @@ -172,3 +173,42 @@ async def test_knob_unset_ignores_terminal_stop_reason(monkeypatch: MonkeyPatch) def test_unknown_exception_name_raises_at_construction(): with pytest.raises(ValueError, match="NotAnException"): Router(model_list=[], treat_finish_reason_as_failure={"x": "NotAnException"}) + + +@pytest.mark.asyncio +async def test_mapped_finish_reason_helpers_direct(monkeypatch: MonkeyPatch): + """Direct coverage of the knob helpers (the router code-coverage check matches by name).""" + fake = FakeAnthropicUpstream() + router = Router( + model_list=[FABLE_TIER, OPUS_TARGET], + treat_finish_reason_as_failure=_knob(), + default_fallbacks=["opus-target"], + num_retries=0, + allowed_fails=0, + cooldown_time=10, + ) + fake.install(monkeypatch) + + ok = await router.acompletion(model="opus-target", max_tokens=16, messages=[{"role": "user", "content": "hi"}]) + assert router._get_mapped_finish_reason(ok) is None + assert router._generic_fallback_available("fable-tier", {}) is True + + error = router._finish_reason_failure_error(model="fable-tier", reason="model_context_window_exceeded") + assert error.status_code == 429 + + deployment = router.model_list[0] + accounted = router._account_mapped_finish_reason_failure( + model="fable-tier", + deployment=deployment, + reason="model_context_window_exceeded", + kwargs={}, + ) + assert accounted is not None + + with pytest.raises(litellm.RateLimitError): + router._handle_mapped_finish_reason_failure( + model="fable-tier", + deployment=deployment, + reason="model_context_window_exceeded", + kwargs={}, + ) From 8c59d2550b39b2212e5e4c8526a1e5ba92469031 Mon Sep 17 00:00:00 2001 From: Paolo Antinori Date: Fri, 18 Sep 2026 19:56:59 +0200 Subject: [PATCH 10/10] fix(router): narrow the optional map before subscripting; cover the warning path --- litellm/router.py | 7 +++++-- .../router_unit_tests/test_router_finish_reason_failure.py | 6 ++++++ 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 87d3d95d646..7857eb70d92 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8708,9 +8708,12 @@ class Router: def _finish_reason_failure_error(self, model: str, reason: str) -> Exception: """Build the exception instance configured for a mapped finish reason.""" - exception_name: Final = self.treat_finish_reason_as_failure[reason] - exception_cls: Final = getattr(litellm, exception_name) message: Final = f"Response finished with reason '{reason}' (treat_finish_reason_as_failure)." + finish_reason_map: Final = self.treat_finish_reason_as_failure + if finish_reason_map is None: + return litellm.APIError(status_code=500, message=message, llm_provider="", model=model) + exception_name: Final = finish_reason_map[reason] + exception_cls: Final = getattr(litellm, exception_name) if exception_name == "APIError": return exception_cls(status_code=500, message=message, llm_provider="", model=model) return exception_cls(message=message, llm_provider="", model=model) diff --git a/tests/router_unit_tests/test_router_finish_reason_failure.py b/tests/router_unit_tests/test_router_finish_reason_failure.py index 5c9fd639313..17e728e992e 100644 --- a/tests/router_unit_tests/test_router_finish_reason_failure.py +++ b/tests/router_unit_tests/test_router_finish_reason_failure.py @@ -175,6 +175,12 @@ def test_unknown_exception_name_raises_at_construction(): Router(model_list=[], treat_finish_reason_as_failure={"x": "NotAnException"}) +def test_healthy_terminal_key_warns_at_construction(capsys: pytest.CaptureFixture): + Router(model_list=[], treat_finish_reason_as_failure={"stop": "RateLimitError"}) + logged = capsys.readouterr().err + capsys.readouterr().out + assert "healthy terminal reasons" in logged + + @pytest.mark.asyncio async def test_mapped_finish_reason_helpers_direct(monkeypatch: MonkeyPatch): """Direct coverage of the knob helpers (the router code-coverage check matches by name)."""