diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 9d1e921cce4..9050a9405c8 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -68,10 +68,23 @@ def _responses_mode_is_lost_by_prefix_strip( ) +def _points_at_openai_backend(api_base: str | None) -> bool: + """Whether api_base is unset or targets api.openai.com rather than a self-hosted backend. + + The ``openai/`` prefix is also how OpenAI-compatible servers (vLLM, llama.cpp, SGLang, ...) + are declared. Those set a custom api_base and do not implement the Responses API, so bridging + /v1/messages to it there breaks multimodal requests. + """ + if not api_base: + return True + return "api.openai.com" in api_base + + def _should_route_to_responses_api( custom_llm_provider: str | None, requested_model: str | None = None, resolved_model: str | None = None, + api_base: str | None = None, ) -> bool: """Return True when the request should use the Responses API path. @@ -81,7 +94,7 @@ def _should_route_to_responses_api( if litellm.use_chat_completions_url_for_anthropic_messages: return False if custom_llm_provider in _RESPONSES_API_PROVIDERS: - return True + return _points_at_openai_backend(api_base) if custom_llm_provider is None or requested_model is None or resolved_model is None: return False return _responses_mode_is_lost_by_prefix_strip(requested_model, resolved_model, custom_llm_provider) @@ -578,7 +591,7 @@ def anthropic_messages_handler( ) if anthropic_messages_provider_config is None: # Route to Responses API for OpenAI / Azure, chat/completions for everything else. - if _should_route_to_responses_api(custom_llm_provider, original_model, model): + if _should_route_to_responses_api(custom_llm_provider, original_model, model, dynamic_api_base or api_base): return LiteLLMMessagesToResponsesAPIHandler.anthropic_messages_handler( max_tokens=max_tokens, messages=messages, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 01f7a2fb7ab..9f73fdfed15 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -137,6 +137,27 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an assert mock_completion.call_args.kwargs["custom_key"] == "custom_value" +@pytest.mark.parametrize( + "api_base, expected", + [ + (None, True), + ("https://api.openai.com/v1", True), + ("http://localhost:8000/v1", False), + ("http://vllm-host:8000/v1", False), + ], +) +def test_should_route_to_responses_api_considers_api_base_for_openai(api_base, expected): + """Regression test for #40780. A self-hosted OpenAI-compatible backend declared as + ``openai/`` with a custom api_base must not be routed to the OpenAI Responses API + (which reshapes images into ``input_image`` items the backend rejects); only real OpenAI + (api_base unset or api.openai.com) keeps the Responses API path.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + _should_route_to_responses_api, + ) + + assert _should_route_to_responses_api("openai", "openai/model", "model", api_base) is expected + + @pytest.mark.asyncio async def test_anthropic_messages_sanitizes_empty_text_blocks_before_dispatch(): """Regression test for #22930. The unified /v1/messages path must