From c92170cd5991b47be093c02b364b30dfac82db84 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 29 Jul 2026 07:16:10 +0000 Subject: [PATCH] fix(anthropic): return max_tokens stop_reason instead of 400 for tiny max_tokens on reasoning models --- .../adapters/handler.py | 51 ++++++- .../test_handler_max_tokens_truncation.py | 127 ++++++++++++++++++ 2 files changed, 175 insertions(+), 3 deletions(-) create mode 100644 tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_max_tokens_truncation.py diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index 7299fc16897..f98eea3c562 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -26,10 +26,11 @@ from litellm.llms.anthropic.experimental_pass_through.context_management import from litellm.llms.anthropic.experimental_pass_through.utils import ( is_reasoning_auto_summary_enabled, ) +from litellm.llms.base_llm.base_model_iterator import MockResponseIterator from litellm.types.llms.anthropic_messages.anthropic_response import ( AnthropicMessagesResponse, ) -from litellm.types.utils import ModelResponse +from litellm.types.utils import Choices, Message, ModelResponse, Usage from litellm.utils import get_model_info if TYPE_CHECKING: @@ -39,6 +40,34 @@ if TYPE_CHECKING: # Anthropic-only keys already mapped by the translator; strip on extra_kwargs re-merge. ANTHROPIC_ONLY_REQUEST_KEYS: frozenset[str] = frozenset({"output_config"}) +_OUTPUT_TOKEN_LIMIT_ERROR_MARKER = "max_tokens or model output limit was reached" + + +def _is_output_token_limit_error(exc: Exception) -> bool: + """True for the provider 400 raised when the output-token budget is too + small to finish even one token (e.g. OpenAI GPT-5.x with ``max_tokens=1``).""" + if not isinstance(exc, litellm.BadRequestError): + return False + message = getattr(exc, "message", None) or str(exc) + return _OUTPUT_TOKEN_LIMIT_ERROR_MARKER in message.lower() + + +def _build_max_tokens_truncation_response(model: str) -> ModelResponse: + """Synthesize an empty ``finish_reason="length"`` response so the empty, + ``max_tokens``-truncated turn flows through the same translation path as a + real completion (mapping to Anthropic ``stop_reason="max_tokens"``).""" + return ModelResponse( + model=model, + choices=[ + Choices( + index=0, + finish_reason="length", + message=Message(role="assistant", content=""), + ) + ], + usage=Usage(prompt_tokens=0, completion_tokens=0, total_tokens=0), + ) + def _messages_have_compaction_block(messages: List[Dict]) -> bool: """Return True when any message carries a ``compaction`` content block.""" @@ -597,7 +626,15 @@ class LiteLLMMessagesToCompletionTransformationHandler: extra_kwargs=kwargs, ) - completion_response = await litellm.acompletion(**completion_kwargs) + try: + completion_response = await litellm.acompletion(**completion_kwargs) + except Exception as e: + if not _is_output_token_limit_error(e): + raise + truncation_response = _build_max_tokens_truncation_response(model) + completion_response = ( + MockResponseIterator(model_response=truncation_response) if stream else truncation_response + ) if stream: transformed_stream = ANTHROPIC_ADAPTER.translate_completion_output_params_streaming( @@ -738,7 +775,15 @@ class LiteLLMMessagesToCompletionTransformationHandler: extra_kwargs=kwargs, ) - completion_response = litellm.completion(**completion_kwargs) + try: + completion_response = litellm.completion(**completion_kwargs) + except Exception as e: + if not _is_output_token_limit_error(e): + raise + truncation_response = _build_max_tokens_truncation_response(model) + completion_response = ( + MockResponseIterator(model_response=truncation_response) if stream else truncation_response + ) if stream: transformed_stream = ANTHROPIC_ADAPTER.translate_completion_output_params_streaming( diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_max_tokens_truncation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_max_tokens_truncation.py new file mode 100644 index 00000000000..c75d7a8dce9 --- /dev/null +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_max_tokens_truncation.py @@ -0,0 +1,127 @@ +""" +Regression tests for issue #35061: Anthropic ``/v1/messages`` with a tiny +``max_tokens`` (e.g. Claude Code's ``max_tokens=1`` model-availability probe) +routed to an OpenAI-family reasoning model (GPT-5.x). + +What was broken: +* The reasoning model can't finish even one output token within such a small + budget, so the provider returns a 400 whose message contains + "Could not finish the message because max_tokens or model output limit was + reached". The adapter surfaced that as a ``BadRequestError``, which made + Claude Code believe the model was unavailable. +* Anthropic's own Messages API returns a 200 with ``stop_reason="max_tokens"`` + for the same request, so the gateway must mirror that contract. + +These tests drive the handler with ``litellm.(a)completion`` mocked to raise the +provider error, so they fail on the pre-fix code (which re-raised) and pass only +once the handler translates the error into a ``max_tokens`` response. An +unrelated ``BadRequestError`` must still propagate untouched. +""" + +import os +import sys + +import pytest + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../.."))) + +import litellm +from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( + LiteLLMMessagesToCompletionTransformationHandler, +) + +MESSAGES = [{"role": "user", "content": "hello"}] +MODEL = "gpt-5.2" +OUTPUT_LIMIT_MESSAGE = ( + "Could not finish the message because max_tokens or model output limit " + "was reached. Please try again with higher max_tokens." +) + + +def _output_limit_error() -> litellm.BadRequestError: + return litellm.BadRequestError(message=OUTPUT_LIMIT_MESSAGE, model=MODEL, llm_provider="openai") + + +def _unrelated_bad_request() -> litellm.BadRequestError: + return litellm.BadRequestError(message="Invalid 'temperature': must be <= 2", model=MODEL, llm_provider="openai") + + +async def _collect(stream) -> bytes: + chunks = [] + async for chunk in stream: + chunks.append(chunk) + return b"".join(chunks) + + +@pytest.mark.asyncio +async def test_async_non_streaming_returns_max_tokens_response(monkeypatch): + async def _raise(**_kwargs): + raise _output_limit_error() + + monkeypatch.setattr(litellm, "acompletion", _raise) + + response = await LiteLLMMessagesToCompletionTransformationHandler.async_anthropic_messages_handler( + max_tokens=1, messages=MESSAGES, model=MODEL + ) + + assert response["stop_reason"] == "max_tokens" + assert response["type"] == "message" + assert response["role"] == "assistant" + # No tokens were produced (or billed) for the rejected request. + assert response["usage"]["output_tokens"] == 0 + + +@pytest.mark.asyncio +async def test_async_streaming_returns_max_tokens_sse(monkeypatch): + async def _raise(**_kwargs): + raise _output_limit_error() + + monkeypatch.setattr(litellm, "acompletion", _raise) + + stream = await LiteLLMMessagesToCompletionTransformationHandler.async_anthropic_messages_handler( + max_tokens=1, messages=MESSAGES, model=MODEL, stream=True + ) + sse = await _collect(stream) + + assert b"event: message_start" in sse + assert b"event: message_stop" in sse + assert b'"stop_reason": "max_tokens"' in sse + + +@pytest.mark.asyncio +async def test_async_unrelated_bad_request_still_raises(monkeypatch): + async def _raise(**_kwargs): + raise _unrelated_bad_request() + + monkeypatch.setattr(litellm, "acompletion", _raise) + + with pytest.raises(litellm.BadRequestError): + await LiteLLMMessagesToCompletionTransformationHandler.async_anthropic_messages_handler( + max_tokens=1, messages=MESSAGES, model=MODEL + ) + + +def test_sync_non_streaming_returns_max_tokens_response(monkeypatch): + def _raise(**_kwargs): + raise _output_limit_error() + + monkeypatch.setattr(litellm, "completion", _raise) + + response = LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler( + max_tokens=1, messages=MESSAGES, model=MODEL, _is_async=False + ) + + assert response["stop_reason"] == "max_tokens" + assert response["usage"]["output_tokens"] == 0 + + +def test_sync_unrelated_bad_request_still_raises(monkeypatch): + def _raise(**_kwargs): + raise _unrelated_bad_request() + + monkeypatch.setattr(litellm, "completion", _raise) + + with pytest.raises(litellm.BadRequestError): + LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler( + max_tokens=1, messages=MESSAGES, model=MODEL, _is_async=False + )