fix(anthropic): route custom-base openai/ through chat/completions on /v1/messages

openai/ with a custom api_base is a self-hosted OpenAI-compatible backend, so bridge /v1/messages through chat/completions instead of the Responses API. Real OpenAI (api_base unset or api.openai.com) is unchanged.
This commit is contained in:
Sameer 2026-09-12 16:25:49 +05:30
parent 9a715df212
commit 8b00177ce5
2 changed files with 36 additions and 2 deletions

View file

@ -68,10 +68,23 @@ def _responses_mode_is_lost_by_prefix_strip(
)
def _points_at_openai_backend(api_base: str | None) -> bool:
"""Whether api_base is unset or targets api.openai.com rather than a self-hosted backend.
The ``openai/`` prefix is also how OpenAI-compatible servers (vLLM, llama.cpp, SGLang, ...)
are declared. Those set a custom api_base and do not implement the Responses API, so bridging
/v1/messages to it there breaks multimodal requests.
"""
if not api_base:
return True
return "api.openai.com" in api_base
def _should_route_to_responses_api(
custom_llm_provider: str | None,
requested_model: str | None = None,
resolved_model: str | None = None,
api_base: str | None = None,
) -> bool:
"""Return True when the request should use the Responses API path.
@ -81,7 +94,7 @@ def _should_route_to_responses_api(
if litellm.use_chat_completions_url_for_anthropic_messages:
return False
if custom_llm_provider in _RESPONSES_API_PROVIDERS:
return True
return _points_at_openai_backend(api_base)
if custom_llm_provider is None or requested_model is None or resolved_model is None:
return False
return _responses_mode_is_lost_by_prefix_strip(requested_model, resolved_model, custom_llm_provider)
@ -578,7 +591,7 @@ def anthropic_messages_handler(
)
if anthropic_messages_provider_config is None:
# Route to Responses API for OpenAI / Azure, chat/completions for everything else.
if _should_route_to_responses_api(custom_llm_provider, original_model, model):
if _should_route_to_responses_api(custom_llm_provider, original_model, model, dynamic_api_base or api_base):
return LiteLLMMessagesToResponsesAPIHandler.anthropic_messages_handler(
max_tokens=max_tokens,
messages=messages,

View file

@ -137,6 +137,27 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an
assert mock_completion.call_args.kwargs["custom_key"] == "custom_value"
@pytest.mark.parametrize(
"api_base, expected",
[
(None, True),
("https://api.openai.com/v1", True),
("http://localhost:8000/v1", False),
("http://vllm-host:8000/v1", False),
],
)
def test_should_route_to_responses_api_considers_api_base_for_openai(api_base, expected):
"""Regression test for #40780. A self-hosted OpenAI-compatible backend declared as
``openai/<model>`` with a custom api_base must not be routed to the OpenAI Responses API
(which reshapes images into ``input_image`` items the backend rejects); only real OpenAI
(api_base unset or api.openai.com) keeps the Responses API path."""
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
_should_route_to_responses_api,
)
assert _should_route_to_responses_api("openai", "openai/model", "model", api_base) is expected
@pytest.mark.asyncio
async def test_anthropic_messages_sanitizes_empty_text_blocks_before_dispatch():
"""Regression test for #22930. The unified /v1/messages path must