From 8b00177ce58d1d40694f858828c4ed2864f4aac4 Mon Sep 17 00:00:00 2001 From: Sameer <121498481+sam7eer@users.noreply.github.com> Date: Sat, 12 Sep 2026 16:25:49 +0530 Subject: [PATCH 1/3] fix(anthropic): route custom-base openai/ through chat/completions on /v1/messages openai/ with a custom api_base is a self-hosted OpenAI-compatible backend, so bridge /v1/messages through chat/completions instead of the Responses API. Real OpenAI (api_base unset or api.openai.com) is unchanged. --- .../messages/handler.py | 17 +++++++++++++-- ...erimental_pass_through_messages_handler.py | 21 +++++++++++++++++++ 2 files changed, 36 insertions(+), 2 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 9d1e921cce4..9050a9405c8 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -68,10 +68,23 @@ def _responses_mode_is_lost_by_prefix_strip( ) +def _points_at_openai_backend(api_base: str | None) -> bool: + """Whether api_base is unset or targets api.openai.com rather than a self-hosted backend. + + The ``openai/`` prefix is also how OpenAI-compatible servers (vLLM, llama.cpp, SGLang, ...) + are declared. Those set a custom api_base and do not implement the Responses API, so bridging + /v1/messages to it there breaks multimodal requests. + """ + if not api_base: + return True + return "api.openai.com" in api_base + + def _should_route_to_responses_api( custom_llm_provider: str | None, requested_model: str | None = None, resolved_model: str | None = None, + api_base: str | None = None, ) -> bool: """Return True when the request should use the Responses API path. @@ -81,7 +94,7 @@ def _should_route_to_responses_api( if litellm.use_chat_completions_url_for_anthropic_messages: return False if custom_llm_provider in _RESPONSES_API_PROVIDERS: - return True + return _points_at_openai_backend(api_base) if custom_llm_provider is None or requested_model is None or resolved_model is None: return False return _responses_mode_is_lost_by_prefix_strip(requested_model, resolved_model, custom_llm_provider) @@ -578,7 +591,7 @@ def anthropic_messages_handler( ) if anthropic_messages_provider_config is None: # Route to Responses API for OpenAI / Azure, chat/completions for everything else. - if _should_route_to_responses_api(custom_llm_provider, original_model, model): + if _should_route_to_responses_api(custom_llm_provider, original_model, model, dynamic_api_base or api_base): return LiteLLMMessagesToResponsesAPIHandler.anthropic_messages_handler( max_tokens=max_tokens, messages=messages, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 01f7a2fb7ab..9f73fdfed15 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -137,6 +137,27 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an assert mock_completion.call_args.kwargs["custom_key"] == "custom_value" +@pytest.mark.parametrize( + "api_base, expected", + [ + (None, True), + ("https://api.openai.com/v1", True), + ("http://localhost:8000/v1", False), + ("http://vllm-host:8000/v1", False), + ], +) +def test_should_route_to_responses_api_considers_api_base_for_openai(api_base, expected): + """Regression test for #40780. A self-hosted OpenAI-compatible backend declared as + ``openai/`` with a custom api_base must not be routed to the OpenAI Responses API + (which reshapes images into ``input_image`` items the backend rejects); only real OpenAI + (api_base unset or api.openai.com) keeps the Responses API path.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + _should_route_to_responses_api, + ) + + assert _should_route_to_responses_api("openai", "openai/model", "model", api_base) is expected + + @pytest.mark.asyncio async def test_anthropic_messages_sanitizes_empty_text_blocks_before_dispatch(): """Regression test for #22930. The unified /v1/messages path must From 3bcc4b68ed5ec75d0b895a404b5fa398c2a93d79 Mon Sep 17 00:00:00 2001 From: Sameer <121498481+sam7eer@users.noreply.github.com> Date: Sat, 12 Sep 2026 18:45:41 +0530 Subject: [PATCH 2/3] fix(anthropic): match api_base by hostname and add request-path regression test --- .../messages/handler.py | 5 ++- ...erimental_pass_through_messages_handler.py | 32 ++++++++++++++++++- 2 files changed, 35 insertions(+), 2 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 9050a9405c8..fd6f9f1db07 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -77,7 +77,10 @@ def _points_at_openai_backend(api_base: str | None) -> bool: """ if not api_base: return True - return "api.openai.com" in api_base + from urllib.parse import urlparse + + parsed = urlparse(api_base if "://" in api_base else f"//{api_base}") + return (parsed.hostname or "").lower() == "api.openai.com" def _should_route_to_responses_api( diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 9f73fdfed15..1d1867d99d4 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -142,15 +142,21 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an [ (None, True), ("https://api.openai.com/v1", True), + ("api.openai.com", True), + ("HTTPS://API.OPENAI.COM/v1", True), ("http://localhost:8000/v1", False), ("http://vllm-host:8000/v1", False), + ("https://api.openai.com.evil.example/v1", False), + ("https://not-api.openai.com.internal/v1", False), ], ) def test_should_route_to_responses_api_considers_api_base_for_openai(api_base, expected): """Regression test for #40780. A self-hosted OpenAI-compatible backend declared as ``openai/`` with a custom api_base must not be routed to the OpenAI Responses API (which reshapes images into ``input_image`` items the backend rejects); only real OpenAI - (api_base unset or api.openai.com) keeps the Responses API path.""" + (api_base unset or the api.openai.com host) keeps the Responses API path. Hostname matching + is normalized, so case variants resolve to OpenAI while lookalike hosts that merely contain + the string do not.""" from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( _should_route_to_responses_api, ) @@ -158,6 +164,30 @@ def test_should_route_to_responses_api_considers_api_base_for_openai(api_base, e assert _should_route_to_responses_api("openai", "openai/model", "model", api_base) is expected +def test_openai_custom_api_base_forwards_messages_to_chat_completions(): + """Regression test for #40780 at the request-path level: driving the real handler, an + ``openai/`` deployment with a custom api_base must forward /v1/messages to chat/completions + (``litellm.completion``) rather than the Responses API, guarding the api_base wiring at the + call site, not just the routing helper.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + anthropic_messages_handler, + ) + + with patch("litellm.completion") as mock_completion: # test-quality-ok: routes via real handler; no injection seam + try: + anthropic_messages_handler( + max_tokens=100, + messages=[{"role": "user", "content": "Hello, how are you?"}], + model="openai/my-local-model", + api_base="http://localhost:8000/v1", + api_key="sk-noauth", + ) + except (ValueError, TypeError, AttributeError) as e: + print(f"Error: {e}") + mock_completion.assert_called_once() + assert mock_completion.call_args.kwargs["api_base"] == "http://localhost:8000/v1" + + @pytest.mark.asyncio async def test_anthropic_messages_sanitizes_empty_text_blocks_before_dispatch(): """Regression test for #22930. The unified /v1/messages path must From f73ec1875cc90bd4dfef060a26ac329d864163c7 Mon Sep 17 00:00:00 2001 From: Sameer <121498481+sam7eer@users.noreply.github.com> Date: Sun, 27 Sep 2026 17:30:04 +0530 Subject: [PATCH 3/3] test(anthropic): move the TQ002 suppression onto the reported def line --- ...st_anthropic_experimental_pass_through_messages_handler.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 92518f5378d..101ca6e896d 100644 --- a/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -160,13 +160,13 @@ def test_should_route_to_responses_api_considers_api_base_for_openai(api_base, e assert _should_route_to_responses_api("openai", "openai/model", "model", api_base) is expected -def test_openai_custom_api_base_forwards_messages_to_chat_completions(): +def test_openai_custom_api_base_forwards_messages_to_chat_completions(): # test-quality-ok: routing guard, asserts /v1/messages reaches litellm.completion with api_base forwarded; the routing decision itself is behaviorally covered by test_should_route_to_responses_api_considers_api_base_for_openai """Regression test for #40780: the full handler forwards a custom-api_base openai/ request to litellm.completion.""" from litellm.llms.anthropic.pass_through.messages.handler import ( anthropic_messages_handler, ) - with patch("litellm.completion") as mock_completion: # test-quality-ok: routes via real handler; no injection seam + with patch("litellm.completion") as mock_completion: try: anthropic_messages_handler( max_tokens=100,