mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(anthropic): route custom-base openai/ through chat/completions on /v1/messages
openai/ with a custom api_base is a self-hosted OpenAI-compatible backend, so bridge /v1/messages through chat/completions instead of the Responses API. Real OpenAI (api_base unset or api.openai.com) is unchanged.
This commit is contained in:
parent
9a715df212
commit
8b00177ce5
2 changed files with 36 additions and 2 deletions
|
|
@ -68,10 +68,23 @@ def _responses_mode_is_lost_by_prefix_strip(
|
|||
)
|
||||
|
||||
|
||||
def _points_at_openai_backend(api_base: str | None) -> bool:
|
||||
"""Whether api_base is unset or targets api.openai.com rather than a self-hosted backend.
|
||||
|
||||
The ``openai/`` prefix is also how OpenAI-compatible servers (vLLM, llama.cpp, SGLang, ...)
|
||||
are declared. Those set a custom api_base and do not implement the Responses API, so bridging
|
||||
/v1/messages to it there breaks multimodal requests.
|
||||
"""
|
||||
if not api_base:
|
||||
return True
|
||||
return "api.openai.com" in api_base
|
||||
|
||||
|
||||
def _should_route_to_responses_api(
|
||||
custom_llm_provider: str | None,
|
||||
requested_model: str | None = None,
|
||||
resolved_model: str | None = None,
|
||||
api_base: str | None = None,
|
||||
) -> bool:
|
||||
"""Return True when the request should use the Responses API path.
|
||||
|
||||
|
|
@ -81,7 +94,7 @@ def _should_route_to_responses_api(
|
|||
if litellm.use_chat_completions_url_for_anthropic_messages:
|
||||
return False
|
||||
if custom_llm_provider in _RESPONSES_API_PROVIDERS:
|
||||
return True
|
||||
return _points_at_openai_backend(api_base)
|
||||
if custom_llm_provider is None or requested_model is None or resolved_model is None:
|
||||
return False
|
||||
return _responses_mode_is_lost_by_prefix_strip(requested_model, resolved_model, custom_llm_provider)
|
||||
|
|
@ -578,7 +591,7 @@ def anthropic_messages_handler(
|
|||
)
|
||||
if anthropic_messages_provider_config is None:
|
||||
# Route to Responses API for OpenAI / Azure, chat/completions for everything else.
|
||||
if _should_route_to_responses_api(custom_llm_provider, original_model, model):
|
||||
if _should_route_to_responses_api(custom_llm_provider, original_model, model, dynamic_api_base or api_base):
|
||||
return LiteLLMMessagesToResponsesAPIHandler.anthropic_messages_handler(
|
||||
max_tokens=max_tokens,
|
||||
messages=messages,
|
||||
|
|
|
|||
|
|
@ -137,6 +137,27 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an
|
|||
assert mock_completion.call_args.kwargs["custom_key"] == "custom_value"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"api_base, expected",
|
||||
[
|
||||
(None, True),
|
||||
("https://api.openai.com/v1", True),
|
||||
("http://localhost:8000/v1", False),
|
||||
("http://vllm-host:8000/v1", False),
|
||||
],
|
||||
)
|
||||
def test_should_route_to_responses_api_considers_api_base_for_openai(api_base, expected):
|
||||
"""Regression test for #40780. A self-hosted OpenAI-compatible backend declared as
|
||||
``openai/<model>`` with a custom api_base must not be routed to the OpenAI Responses API
|
||||
(which reshapes images into ``input_image`` items the backend rejects); only real OpenAI
|
||||
(api_base unset or api.openai.com) keeps the Responses API path."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
|
||||
_should_route_to_responses_api,
|
||||
)
|
||||
|
||||
assert _should_route_to_responses_api("openai", "openai/model", "model", api_base) is expected
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_messages_sanitizes_empty_text_blocks_before_dispatch():
|
||||
"""Regression test for #22930. The unified /v1/messages path must
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue