diff --git a/litellm/llms/anthropic/pass_through/messages/handler.py b/litellm/llms/anthropic/pass_through/messages/handler.py index 3068a7eb3ab..62e1dceea23 100644 --- a/litellm/llms/anthropic/pass_through/messages/handler.py +++ b/litellm/llms/anthropic/pass_through/messages/handler.py @@ -70,10 +70,22 @@ def _responses_mode_is_lost_by_prefix_strip( ) +def _points_at_openai_backend(api_base: str | None) -> bool: + """Only real OpenAI hosts keep the Responses path; a custom api_base means a self-hosted + openai/-compatible backend (vLLM, llama.cpp, ...) that lacks the Responses API.""" + if not api_base: + return True + from urllib.parse import urlparse + + host = (urlparse(api_base if "://" in api_base else f"//{api_base}").hostname or "").lower() + return host == "api.openai.com" or host.endswith(".openai.com") + + def _should_route_to_responses_api( custom_llm_provider: str | None, requested_model: str | None = None, resolved_model: str | None = None, + api_base: str | None = None, ) -> bool: """Return True when the request should use the Responses API path. @@ -83,7 +95,7 @@ def _should_route_to_responses_api( if litellm.use_chat_completions_url_for_anthropic_messages: return False if custom_llm_provider in _RESPONSES_API_PROVIDERS: - return True + return _points_at_openai_backend(api_base) if custom_llm_provider is None or requested_model is None or resolved_model is None: return False return _responses_mode_is_lost_by_prefix_strip(requested_model, resolved_model, custom_llm_provider) @@ -581,7 +593,7 @@ def anthropic_messages_handler( if anthropic_messages_provider_config is None: # Route to Responses API for OpenAI / Azure, chat/completions for everything else. if kwargs.get("compaction") is None and _should_route_to_responses_api( - custom_llm_provider, original_model, model + custom_llm_provider, original_model, model, dynamic_api_base or api_base ): return LiteLLMMessagesToResponsesAPIHandler.anthropic_messages_handler( max_tokens=max_tokens, diff --git a/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index c3d4dba7376..101ca6e896d 100644 --- a/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -137,6 +137,50 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an assert mock_completion.call_args.kwargs["custom_key"] == "custom_value" +@pytest.mark.parametrize( + "api_base, expected", + [ + (None, True), + ("https://api.openai.com/v1", True), + ("api.openai.com", True), + ("HTTPS://API.OPENAI.COM/v1", True), + ("http://localhost:8000/v1", False), + ("http://vllm-host:8000/v1", False), + ("https://my-org.privatelink.openai.com/v1", True), + ("https://api.openai.com.evil.example/v1", False), + ("https://not-api.openai.com.internal/v1", False), + ], +) +def test_should_route_to_responses_api_considers_api_base_for_openai(api_base, expected): + """Regression test for #40780: openai/ with a custom api_base routes to chat/completions, real OpenAI hosts keep Responses.""" + from litellm.llms.anthropic.pass_through.messages.handler import ( + _should_route_to_responses_api, + ) + + assert _should_route_to_responses_api("openai", "openai/model", "model", api_base) is expected + + +def test_openai_custom_api_base_forwards_messages_to_chat_completions(): # test-quality-ok: routing guard, asserts /v1/messages reaches litellm.completion with api_base forwarded; the routing decision itself is behaviorally covered by test_should_route_to_responses_api_considers_api_base_for_openai + """Regression test for #40780: the full handler forwards a custom-api_base openai/ request to litellm.completion.""" + from litellm.llms.anthropic.pass_through.messages.handler import ( + anthropic_messages_handler, + ) + + with patch("litellm.completion") as mock_completion: + try: + anthropic_messages_handler( + max_tokens=100, + messages=[{"role": "user", "content": "Hello, how are you?"}], + model="openai/my-local-model", + api_base="http://localhost:8000/v1", + api_key="sk-noauth", + ) + except (ValueError, TypeError, AttributeError) as e: + print(f"Error: {e}") + mock_completion.assert_called_once() + assert mock_completion.call_args.kwargs["api_base"] == "http://localhost:8000/v1" + + @pytest.mark.asyncio async def test_anthropic_messages_sanitizes_empty_text_blocks_before_dispatch(): """Regression test for #22930. The unified /v1/messages path must