fix(anthropic): return max_tokens stop_reason instead of 400 for tiny max_tokens on reasoning models

This commit is contained in:
Devin AI 2026-07-29 07:16:10 +00:00
parent c274cf321c
commit c92170cd59
2 changed files with 175 additions and 3 deletions

View file

@ -26,10 +26,11 @@ from litellm.llms.anthropic.experimental_pass_through.context_management import
from litellm.llms.anthropic.experimental_pass_through.utils import (
is_reasoning_auto_summary_enabled,
)
from litellm.llms.base_llm.base_model_iterator import MockResponseIterator
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
from litellm.types.utils import ModelResponse
from litellm.types.utils import Choices, Message, ModelResponse, Usage
from litellm.utils import get_model_info
if TYPE_CHECKING:
@ -39,6 +40,34 @@ if TYPE_CHECKING:
# Anthropic-only keys already mapped by the translator; strip on extra_kwargs re-merge.
ANTHROPIC_ONLY_REQUEST_KEYS: frozenset[str] = frozenset({"output_config"})
_OUTPUT_TOKEN_LIMIT_ERROR_MARKER = "max_tokens or model output limit was reached"
def _is_output_token_limit_error(exc: Exception) -> bool:
"""True for the provider 400 raised when the output-token budget is too
small to finish even one token (e.g. OpenAI GPT-5.x with ``max_tokens=1``)."""
if not isinstance(exc, litellm.BadRequestError):
return False
message = getattr(exc, "message", None) or str(exc)
return _OUTPUT_TOKEN_LIMIT_ERROR_MARKER in message.lower()
def _build_max_tokens_truncation_response(model: str) -> ModelResponse:
"""Synthesize an empty ``finish_reason="length"`` response so the empty,
``max_tokens``-truncated turn flows through the same translation path as a
real completion (mapping to Anthropic ``stop_reason="max_tokens"``)."""
return ModelResponse(
model=model,
choices=[
Choices(
index=0,
finish_reason="length",
message=Message(role="assistant", content=""),
)
],
usage=Usage(prompt_tokens=0, completion_tokens=0, total_tokens=0),
)
def _messages_have_compaction_block(messages: List[Dict]) -> bool:
"""Return True when any message carries a ``compaction`` content block."""
@ -597,7 +626,15 @@ class LiteLLMMessagesToCompletionTransformationHandler:
extra_kwargs=kwargs,
)
completion_response = await litellm.acompletion(**completion_kwargs)
try:
completion_response = await litellm.acompletion(**completion_kwargs)
except Exception as e:
if not _is_output_token_limit_error(e):
raise
truncation_response = _build_max_tokens_truncation_response(model)
completion_response = (
MockResponseIterator(model_response=truncation_response) if stream else truncation_response
)
if stream:
transformed_stream = ANTHROPIC_ADAPTER.translate_completion_output_params_streaming(
@ -738,7 +775,15 @@ class LiteLLMMessagesToCompletionTransformationHandler:
extra_kwargs=kwargs,
)
completion_response = litellm.completion(**completion_kwargs)
try:
completion_response = litellm.completion(**completion_kwargs)
except Exception as e:
if not _is_output_token_limit_error(e):
raise
truncation_response = _build_max_tokens_truncation_response(model)
completion_response = (
MockResponseIterator(model_response=truncation_response) if stream else truncation_response
)
if stream:
transformed_stream = ANTHROPIC_ADAPTER.translate_completion_output_params_streaming(

View file

@ -0,0 +1,127 @@
"""
Regression tests for issue #35061: Anthropic ``/v1/messages`` with a tiny
``max_tokens`` (e.g. Claude Code's ``max_tokens=1`` model-availability probe)
routed to an OpenAI-family reasoning model (GPT-5.x).
What was broken:
* The reasoning model can't finish even one output token within such a small
budget, so the provider returns a 400 whose message contains
"Could not finish the message because max_tokens or model output limit was
reached". The adapter surfaced that as a ``BadRequestError``, which made
Claude Code believe the model was unavailable.
* Anthropic's own Messages API returns a 200 with ``stop_reason="max_tokens"``
for the same request, so the gateway must mirror that contract.
These tests drive the handler with ``litellm.(a)completion`` mocked to raise the
provider error, so they fail on the pre-fix code (which re-raised) and pass only
once the handler translates the error into a ``max_tokens`` response. An
unrelated ``BadRequestError`` must still propagate untouched.
"""
import os
import sys
import pytest
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../..")))
import litellm
from litellm.llms.anthropic.experimental_pass_through.adapters.handler import (
LiteLLMMessagesToCompletionTransformationHandler,
)
MESSAGES = [{"role": "user", "content": "hello"}]
MODEL = "gpt-5.2"
OUTPUT_LIMIT_MESSAGE = (
"Could not finish the message because max_tokens or model output limit "
"was reached. Please try again with higher max_tokens."
)
def _output_limit_error() -> litellm.BadRequestError:
return litellm.BadRequestError(message=OUTPUT_LIMIT_MESSAGE, model=MODEL, llm_provider="openai")
def _unrelated_bad_request() -> litellm.BadRequestError:
return litellm.BadRequestError(message="Invalid 'temperature': must be <= 2", model=MODEL, llm_provider="openai")
async def _collect(stream) -> bytes:
chunks = []
async for chunk in stream:
chunks.append(chunk)
return b"".join(chunks)
@pytest.mark.asyncio
async def test_async_non_streaming_returns_max_tokens_response(monkeypatch):
async def _raise(**_kwargs):
raise _output_limit_error()
monkeypatch.setattr(litellm, "acompletion", _raise)
response = await LiteLLMMessagesToCompletionTransformationHandler.async_anthropic_messages_handler(
max_tokens=1, messages=MESSAGES, model=MODEL
)
assert response["stop_reason"] == "max_tokens"
assert response["type"] == "message"
assert response["role"] == "assistant"
# No tokens were produced (or billed) for the rejected request.
assert response["usage"]["output_tokens"] == 0
@pytest.mark.asyncio
async def test_async_streaming_returns_max_tokens_sse(monkeypatch):
async def _raise(**_kwargs):
raise _output_limit_error()
monkeypatch.setattr(litellm, "acompletion", _raise)
stream = await LiteLLMMessagesToCompletionTransformationHandler.async_anthropic_messages_handler(
max_tokens=1, messages=MESSAGES, model=MODEL, stream=True
)
sse = await _collect(stream)
assert b"event: message_start" in sse
assert b"event: message_stop" in sse
assert b'"stop_reason": "max_tokens"' in sse
@pytest.mark.asyncio
async def test_async_unrelated_bad_request_still_raises(monkeypatch):
async def _raise(**_kwargs):
raise _unrelated_bad_request()
monkeypatch.setattr(litellm, "acompletion", _raise)
with pytest.raises(litellm.BadRequestError):
await LiteLLMMessagesToCompletionTransformationHandler.async_anthropic_messages_handler(
max_tokens=1, messages=MESSAGES, model=MODEL
)
def test_sync_non_streaming_returns_max_tokens_response(monkeypatch):
def _raise(**_kwargs):
raise _output_limit_error()
monkeypatch.setattr(litellm, "completion", _raise)
response = LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler(
max_tokens=1, messages=MESSAGES, model=MODEL, _is_async=False
)
assert response["stop_reason"] == "max_tokens"
assert response["usage"]["output_tokens"] == 0
def test_sync_unrelated_bad_request_still_raises(monkeypatch):
def _raise(**_kwargs):
raise _unrelated_bad_request()
monkeypatch.setattr(litellm, "completion", _raise)
with pytest.raises(litellm.BadRequestError):
LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler(
max_tokens=1, messages=MESSAGES, model=MODEL, _is_async=False
)