diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index c4b5cc628e2..f7716c80c2e 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -42,13 +42,16 @@ from .utils import AnthropicMessagesRequestUtils, mock_response _RESPONSES_API_PROVIDERS: Final = frozenset({"openai"}) -def _should_route_to_responses_api(custom_llm_provider: str | None) -> bool: +def _should_route_to_responses_api( + custom_llm_provider: str | None, + use_chat_completions_api: bool | None = None, +) -> bool: """Return True when the provider should use the Responses API path. Set ``litellm.use_chat_completions_url_for_anthropic_messages = True`` to opt out and route OpenAI/Azure requests through chat/completions instead. """ - if litellm.use_chat_completions_url_for_anthropic_messages: + if litellm.use_chat_completions_url_for_anthropic_messages or use_chat_completions_api is True: return False return custom_llm_provider in _RESPONSES_API_PROVIDERS @@ -551,7 +554,10 @@ def anthropic_messages_handler( custom_llm_provider=custom_llm_provider, **kwargs, ) - if _should_route_to_responses_api(custom_llm_provider): + if _should_route_to_responses_api( + custom_llm_provider=custom_llm_provider, + use_chat_completions_api=kwargs.get("use_chat_completions_api"), + ): return LiteLLMMessagesToResponsesAPIHandler.anthropic_messages_handler(**_shared_kwargs) # The in-gateway context_management polyfill runs inside diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index f11324ca376..0ac051fa612 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -939,6 +939,28 @@ def test_gate_translates_when_supported_endpoints_absent(monkeypatch): assert "config" not in captured +def test_gate_uses_chat_completions_when_requested(monkeypatch): + """The per-deployment chat-completions opt-in must bypass Responses API.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + anthropic_messages_handler, + ) + + captured, translation_calls = _gate_stubs(monkeypatch) + + result = anthropic_messages_handler( + max_tokens=100, + messages=[{"role": "user", "content": "Hello"}], + model="openai/some-model", + api_key="sk-test", + api_base="https://host/v1", + use_chat_completions_api=True, + ) + + assert result == "translated" + assert translation_calls["count"] == 1 + assert "config" not in captured + + def test_gate_passthrough_skipped_when_only_chat_completions_supported(monkeypatch): """A deployment that lists only /v1/chat/completions is still translated; the opt-in is specifically the /v1/messages entry."""