From 6d828e5759dc76ecc5d582861fd56e27a5763c2d Mon Sep 17 00:00:00 2001 From: Mateo Wang <277851410+mateo-berri@users.noreply.github.com> Date: Tue, 30 Jun 2026 12:17:33 -0700 Subject: [PATCH] feat(messages): passthrough /v1/messages to native endpoints via supported_endpoints (#31685) * feat(messages): passthrough /v1/messages to native endpoints via supported_endpoints The unified /v1/messages proxy endpoint always translated inbound Anthropic requests down to /v1/chat/completions (or the Responses API for openai) when the deployment's provider lacked a native Anthropic-messages config, dropping Anthropic-only features like cache_control and thinking. Some customers run OpenAI-compatible servers (self-hosted vLLM, DeepSeek's Anthropic endpoint, etc.) that also natively expose /v1/messages and want the raw Anthropic payload forwarded untranslated, while keeping provider openai so /v1/chat/completions to the same deployment stays native. Opt in per deployment via model_info.supported_endpoints containing /v1/messages. When present, the gate routes to a generic, provider-agnostic OpenAILikeAnthropicMessagesConfig that POSTs the Anthropic payload to {api_base}/v1/messages with Bearer auth, instead of translating. Default behavior is unchanged. Generalizes and supersedes the hosted_vllm-only, env-var-toggled PR #28745. * fix(messages): preserve standard-cased caller headers in native passthrough The OpenAI-like Anthropic passthrough config only checked for lowercase header names before injecting Bearer auth, anthropic-version, and content-type defaults. A caller sending standard-cased Authorization, Anthropic-Version, or Content-Type was treated as missing those headers, so LiteLLM added duplicate lowercase variants and overwrote the caller's credential/version at the HTTP layer. Header presence is now checked case-insensitively and the merge no longer mutates the caller dict. Also moves the feature docs out of the main repo (docs live in litellm-docs). * fix(openai_like/messages): delegate to parent transform and inject anthropic-beta headers The passthrough config bypassed the parent transform and skipped header beta injection. Both gaps cause native /v1/messages features (context management, advisor tool, fast mode, structured outputs, reasoning_effort, advisor stripping) to silently degrade on opted-in deployments. Reuse the parent's pipeline and call _update_headers_with_anthropic_beta after merging defaults * fix: normalize anthropic-beta header key case before beta injection * style: collapse anthropic-beta header normalization to single line ruff format --check requires the comprehension on one line (it fits within the 120 char limit); fixes the lint job failure on the bugbot autofix commit * fix(messages): forward anthropic-beta to native passthrough upstream The shared anthropic_messages HTTP handler ran update_headers_with_filtered_beta with the deployment's custom_llm_provider after validate. For the native /v1/messages passthrough that provider is openai, which has no beta-header mapping, so every anthropic-beta value (caller-supplied or feature-derived for speed/context_management/etc.) was stripped to empty before the upstream request, breaking beta passthrough to the Anthropic-compatible endpoint. Beta filtering only makes sense on cross-provider translation paths where the upstream cannot understand Anthropic betas. Gate it on a new should_filter_anthropic_beta_headers() that defaults to True (bedrock, vertex_ai, native anthropic unchanged) and is overridden to False by OpenAILikeAnthropicMessagesConfig, whose upstream is a native Anthropic endpoint, so betas pass through verbatim. * chore: remove accidentally committed local QA logs and config --------- Co-authored-by: Cursor Agent --- .../messages/handler.py | 21 ++ .../anthropic_messages/transformation.py | 11 + litellm/llms/custom_httpx/llm_http_handler.py | 3 +- litellm/llms/openai_like/messages/__init__.py | 0 .../openai_like/messages/transformation.py | 69 ++++ ...erimental_pass_through_messages_handler.py | 106 ++++++ .../llms/openai_like/messages/__init__.py | 0 ..._like_anthropic_messages_transformation.py | 301 ++++++++++++++++++ 8 files changed, 510 insertions(+), 1 deletion(-) create mode 100644 litellm/llms/openai_like/messages/__init__.py create mode 100644 litellm/llms/openai_like/messages/transformation.py create mode 100644 tests/test_litellm/llms/openai_like/messages/__init__.py create mode 100644 tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 9c9427c7302..547ddd9b8d3 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -61,6 +61,19 @@ def _should_route_to_responses_api(custom_llm_provider: Optional[str]) -> bool: return custom_llm_provider in _RESPONSES_API_PROVIDERS +def _deployment_passes_through_anthropic_messages(model_info: object) -> bool: + """Whether the deployment opted into forwarding /v1/messages untranslated. + + The opt-in is ``model_info.supported_endpoints`` containing ``"/v1/messages"``, + declared per deployment in config.yaml and plumbed here as ``kwargs["model_info"]`` + by the router. + """ + if not isinstance(model_info, dict): + return False + supported_endpoints = model_info.get("supported_endpoints") + return isinstance(supported_endpoints, (list, tuple)) and "/v1/messages" in supported_endpoints + + ####### ENVIRONMENT VARIABLES ################### # Initialize any necessary instances or variables here base_llm_http_handler = BaseLLMHTTPHandler() @@ -456,6 +469,14 @@ def anthropic_messages_handler( model=model, provider=litellm.LlmProviders(custom_llm_provider), ) + if anthropic_messages_provider_config is None and _deployment_passes_through_anthropic_messages( + kwargs.get("model_info") + ): + from litellm.llms.openai_like.messages.transformation import ( + OpenAILikeAnthropicMessagesConfig, + ) + + anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig() if anthropic_messages_provider_config is None: # Route to Responses API for OpenAI / Azure, chat/completions for everything else. _shared_kwargs = dict( diff --git a/litellm/llms/base_llm/anthropic_messages/transformation.py b/litellm/llms/base_llm/anthropic_messages/transformation.py index 7f8403c0223..966995bc571 100644 --- a/litellm/llms/base_llm/anthropic_messages/transformation.py +++ b/litellm/llms/base_llm/anthropic_messages/transformation.py @@ -103,6 +103,17 @@ class BaseAnthropicMessagesConfig(ABC): """ return headers, None + def should_filter_anthropic_beta_headers(self) -> bool: + """ + Whether ``anthropic-beta`` header values should be filtered down to the + ones the routed provider supports before the upstream request. + + Cross-provider translation paths (bedrock, vertex_ai, ...) need this so + unsupported betas are dropped. Configs that forward natively to an + Anthropic-compatible endpoint return False to pass betas through verbatim. + """ + return True + def get_async_streaming_response_iterator( self, model: str, diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 9f18b669124..9bb956d0808 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -1986,7 +1986,8 @@ class BaseLLMHTTPHandler: api_base=api_base, ) - headers = update_headers_with_filtered_beta(headers=headers, provider=custom_llm_provider) + if anthropic_messages_provider_config.should_filter_anthropic_beta_headers(): + headers = update_headers_with_filtered_beta(headers=headers, provider=custom_llm_provider) logging_obj.update_from_kwargs( kwargs=kwargs, diff --git a/litellm/llms/openai_like/messages/__init__.py b/litellm/llms/openai_like/messages/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/litellm/llms/openai_like/messages/transformation.py b/litellm/llms/openai_like/messages/transformation.py new file mode 100644 index 00000000000..0df8c6e830b --- /dev/null +++ b/litellm/llms/openai_like/messages/transformation.py @@ -0,0 +1,69 @@ +from typing import Any, Optional + +from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( + AnthropicMessagesConfig, +) + +DEFAULT_ANTHROPIC_API_VERSION = "2023-06-01" + + +class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): + """ + Forwards Anthropic /v1/messages requests to an OpenAI-compatible server that + also natively exposes the Anthropic Messages API, with no translation. + + Opted into per deployment via ``model_info.supported_endpoints`` containing + ``"/v1/messages"``. The inbound Anthropic payload (system, cache_control, + thinking, tools, ...) is forwarded essentially unchanged to + ``{api_base}/v1/messages``, so Anthropic-only features that the + Anthropic->OpenAI translation would otherwise drop are preserved. Response + parsing and streaming are inherited from the native Anthropic config. + """ + + def validate_anthropic_messages_environment( + self, + headers: dict[str, str], + model: str, + messages: list[Any], + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> tuple[dict[str, str], Optional[str]]: + present = {key.lower() for key in headers} + needs_auth = bool(api_key) and "authorization" not in present and "x-api-key" not in present + defaults: dict[str, str] = { + **({"authorization": f"Bearer {api_key}"} if needs_auth else {}), + **({"anthropic-version": DEFAULT_ANTHROPIC_API_VERSION} if "anthropic-version" not in present else {}), + **({"content-type": "application/json"} if "content-type" not in present else {}), + } + combined = {**headers, **defaults} + normalized = { + ("anthropic-beta" if key.lower() == "anthropic-beta" else key): value for key, value in combined.items() + } + merged = self._update_headers_with_anthropic_beta( + headers=normalized, + optional_params=optional_params, + ) + return merged, api_base + + def should_filter_anthropic_beta_headers(self) -> bool: + return False + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + if not api_base: + raise ValueError("api_base is required to forward Anthropic /v1/messages to a native endpoint") + base = api_base.rstrip("/") + if base.endswith("/v1/messages"): + return base + if base.endswith("/v1"): + base = base[: -len("/v1")] + return f"{base}/v1/messages" diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 7bcaf07c5bb..3327fc39f73 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -715,3 +715,109 @@ async def test_async_wrapper_sets_presanitized_and_sanitizes_once(): assert spy.call_count == 1 assert captured["presanitized"] is True assert [b["type"] for b in captured["messages"][0]["content"]] == ["tool_use"] + + +def _gate_stubs(monkeypatch): + """Patch the gate's downstream dispatch targets so config selection can be + observed without making a network call. + + Returns ``(captured, translation_calls)`` where ``captured["config"]`` is the + provider config handed to the native passthrough path and ``translation_calls`` + counts hits on the Anthropic->OpenAI translation handlers. + """ + from litellm.llms.anthropic.experimental_pass_through.messages import handler + + captured = {} + translation_calls = {"count": 0} + + def fake_native(**kwargs): + captured["config"] = kwargs.get("anthropic_messages_provider_config") + return "native-passthrough" + + def fake_translation(**kwargs): + translation_calls["count"] += 1 + return "translated" + + monkeypatch.setattr(handler.base_llm_http_handler, "anthropic_messages_handler", fake_native) + monkeypatch.setattr( + handler.LiteLLMMessagesToResponsesAPIHandler, + "anthropic_messages_handler", + staticmethod(fake_translation), + ) + monkeypatch.setattr( + handler.LiteLLMMessagesToCompletionTransformationHandler, + "anthropic_messages_handler", + staticmethod(fake_translation), + ) + return captured, translation_calls + + +def test_gate_passthrough_when_supported_endpoints_opts_in(monkeypatch): + """provider=openai + model_info.supported_endpoints containing /v1/messages + must route to the native passthrough config, NOT the translation handlers.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + anthropic_messages_handler, + ) + from litellm.llms.openai_like.messages.transformation import ( + OpenAILikeAnthropicMessagesConfig, + ) + + captured, translation_calls = _gate_stubs(monkeypatch) + + result = anthropic_messages_handler( + max_tokens=100, + messages=[{"role": "user", "content": "Hello"}], + model="openai/some-model", + api_key="sk-test", + api_base="https://host/v1", + model_info={"supported_endpoints": ["/v1/chat/completions", "/v1/messages"]}, + ) + + assert result == "native-passthrough" + assert isinstance(captured["config"], OpenAILikeAnthropicMessagesConfig) + assert translation_calls["count"] == 0 + + +def test_gate_translates_when_supported_endpoints_absent(monkeypatch): + """Default behavior is unchanged: without the /v1/messages opt-in, an openai + deployment is translated (Responses API), never passed through natively.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + anthropic_messages_handler, + ) + + captured, translation_calls = _gate_stubs(monkeypatch) + + result = anthropic_messages_handler( + max_tokens=100, + messages=[{"role": "user", "content": "Hello"}], + model="openai/some-model", + api_key="sk-test", + api_base="https://host/v1", + ) + + assert result == "translated" + assert translation_calls["count"] == 1 + assert "config" not in captured + + +def test_gate_passthrough_skipped_when_only_chat_completions_supported(monkeypatch): + """A deployment that lists only /v1/chat/completions is still translated; + the opt-in is specifically the /v1/messages entry.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + anthropic_messages_handler, + ) + + captured, translation_calls = _gate_stubs(monkeypatch) + + result = anthropic_messages_handler( + max_tokens=100, + messages=[{"role": "user", "content": "Hello"}], + model="openai/some-model", + api_key="sk-test", + api_base="https://host/v1", + model_info={"supported_endpoints": ["/v1/chat/completions"]}, + ) + + assert result == "translated" + assert translation_calls["count"] == 1 + assert "config" not in captured diff --git a/tests/test_litellm/llms/openai_like/messages/__init__.py b/tests/test_litellm/llms/openai_like/messages/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py b/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py new file mode 100644 index 00000000000..534d7aefda4 --- /dev/null +++ b/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py @@ -0,0 +1,301 @@ +import pytest + +from litellm.llms.anthropic.common_utils import AnthropicError +from litellm.llms.openai_like.messages.transformation import ( + OpenAILikeAnthropicMessagesConfig, +) +from litellm.types.router import GenericLiteLLMParams + + +@pytest.fixture +def config() -> OpenAILikeAnthropicMessagesConfig: + return OpenAILikeAnthropicMessagesConfig() + + +@pytest.mark.parametrize( + "api_base, expected", + [ + ("https://host/v1", "https://host/v1/messages"), + ("https://host/v1/", "https://host/v1/messages"), + ("https://host", "https://host/v1/messages"), + ("https://host/v1/messages", "https://host/v1/messages"), + ("https://api.deepseek.com/anthropic", "https://api.deepseek.com/anthropic/v1/messages"), + ("https://api.deepseek.com/anthropic/v1", "https://api.deepseek.com/anthropic/v1/messages"), + ], +) +def test_get_complete_url_handles_api_base_variants(config, api_base, expected): + url = config.get_complete_url( + api_base=api_base, + api_key="sk-test", + model="some-model", + optional_params={}, + litellm_params={}, + ) + assert url == expected + + +def test_get_complete_url_requires_api_base(config): + with pytest.raises(ValueError, match="api_base is required"): + config.get_complete_url( + api_base=None, + api_key="sk-test", + model="some-model", + optional_params={}, + litellm_params={}, + ) + + +def test_request_stays_in_anthropic_shape(config): + messages = [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "Summarize this", + "cache_control": {"type": "ephemeral"}, + } + ], + } + ] + optional_params = { + "max_tokens": 256, + "system": "You are a careful assistant", + "thinking": {"type": "enabled", "budget_tokens": 1024}, + "temperature": 0.3, + "tools": [{"name": "lookup", "input_schema": {"type": "object"}}], + "stream": False, + } + + payload = config.transform_anthropic_messages_request( + model="some-model", + messages=messages, + anthropic_messages_optional_request_params=optional_params, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert payload["model"] == "some-model" + assert payload["messages"] == messages + assert payload["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral"} + assert payload["system"] == "You are a careful assistant" + assert payload["thinking"] == {"type": "enabled", "budget_tokens": 1024} + assert payload["max_tokens"] == 256 + assert payload["tools"] == optional_params["tools"] + + openai_only_keys = { + "max_completion_tokens", + "stop", + "n", + "logprobs", + "response_format", + "frequency_penalty", + } + assert openai_only_keys.isdisjoint(payload.keys()) + + +def test_request_requires_max_tokens(config): + with pytest.raises(AnthropicError, match="max_tokens is required"): + config.transform_anthropic_messages_request( + model="some-model", + messages=[{"role": "user", "content": "hi"}], + anthropic_messages_optional_request_params={"system": "s"}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + +def test_validate_environment_sets_bearer_and_anthropic_defaults(config): + headers, api_base = config.validate_anthropic_messages_environment( + headers={}, + model="some-model", + messages=[], + optional_params={}, + litellm_params={}, + api_key="sk-test", + api_base="https://host/v1", + ) + assert headers["authorization"] == "Bearer sk-test" + assert headers["anthropic-version"] == "2023-06-01" + assert headers["content-type"] == "application/json" + assert api_base == "https://host/v1" + + +def test_validate_environment_does_not_overwrite_caller_headers(config): + headers, _ = config.validate_anthropic_messages_environment( + headers={ + "authorization": "Bearer caller-token", + "anthropic-version": "2024-10-22", + "content-type": "application/json", + }, + model="some-model", + messages=[], + optional_params={}, + litellm_params={}, + api_key="sk-test", + api_base="https://host/v1", + ) + assert headers["authorization"] == "Bearer caller-token" + assert headers["anthropic-version"] == "2024-10-22" + + +def test_validate_environment_preserves_standard_cased_caller_headers(config): + headers, _ = config.validate_anthropic_messages_environment( + headers={ + "Authorization": "Bearer caller-token", + "Anthropic-Version": "2024-10-22", + "Content-Type": "application/json", + }, + model="some-model", + messages=[], + optional_params={}, + litellm_params={}, + api_key="sk-test", + api_base="https://host/v1", + ) + lowercased = {key.lower() for key in headers} + assert len(lowercased) == len(headers) + assert headers["Authorization"] == "Bearer caller-token" + assert headers["Anthropic-Version"] == "2024-10-22" + assert headers["Content-Type"] == "application/json" + + +def test_validate_environment_honors_x_api_key_when_present(config): + headers, _ = config.validate_anthropic_messages_environment( + headers={"X-Api-Key": "caller-key"}, + model="some-model", + messages=[], + optional_params={}, + litellm_params={}, + api_key="sk-test", + api_base="https://host/v1", + ) + assert "authorization" not in {key.lower() for key in headers} + assert headers["X-Api-Key"] == "caller-key" + + +def test_validate_environment_injects_anthropic_beta_for_context_management(config): + headers, _ = config.validate_anthropic_messages_environment( + headers={}, + model="some-model", + messages=[], + optional_params={ + "context_management": {"edits": [{"type": "clear_tool_uses_20250919"}]}, + }, + litellm_params={}, + api_key="sk-test", + api_base="https://host/v1", + ) + assert "context-management-2025-06-27" in headers["anthropic-beta"].split(",") + + +def test_validate_environment_injects_anthropic_beta_for_fast_mode(config): + headers, _ = config.validate_anthropic_messages_environment( + headers={}, + model="some-model", + messages=[], + optional_params={"speed": "fast"}, + litellm_params={}, + api_key="sk-test", + api_base="https://host/v1", + ) + assert "fast-mode-2026-02-01" in headers["anthropic-beta"].split(",") + + +def test_validate_environment_merges_existing_anthropic_beta(config): + headers, _ = config.validate_anthropic_messages_environment( + headers={"anthropic-beta": "caller-flag"}, + model="some-model", + messages=[], + optional_params={"speed": "fast"}, + litellm_params={}, + api_key="sk-test", + api_base="https://host/v1", + ) + beta_values = set(headers["anthropic-beta"].split(",")) + assert "caller-flag" in beta_values + assert "fast-mode-2026-02-01" in beta_values + + +def test_request_strips_advisor_blocks_when_advisor_tool_absent(config): + messages = [ + {"role": "user", "content": "hello"}, + { + "role": "assistant", + "content": [ + {"type": "text", "text": "thinking out loud"}, + {"type": "server_tool_use", "id": "advisor_1", "name": "advisor", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "advisor_1", "content": "stale"}, + ], + }, + ] + + payload = config.transform_anthropic_messages_request( + model="some-model", + messages=messages, + anthropic_messages_optional_request_params={"max_tokens": 64}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + flattened_types = [ + block.get("type") + for message in payload["messages"] + if isinstance(message.get("content"), list) + for block in message["content"] + if isinstance(block, dict) + ] + assert "advisor_tool_result" not in flattened_types + assert "server_tool_use" not in flattened_types + + +def test_request_maps_reasoning_effort_to_thinking(config): + payload = config.transform_anthropic_messages_request( + model="claude-sonnet-4-20250514", + messages=[{"role": "user", "content": "hi"}], + anthropic_messages_optional_request_params={ + "max_tokens": 1024, + "reasoning_effort": "medium", + }, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert "reasoning_effort" not in payload + assert isinstance(payload.get("thinking"), dict) + assert payload["thinking"].get("type") == "enabled" + + +def test_passthrough_disables_anthropic_beta_filtering(config): + from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( + AnthropicMessagesConfig, + ) + + assert config.should_filter_anthropic_beta_headers() is False + assert AnthropicMessagesConfig().should_filter_anthropic_beta_headers() is True + + +def test_anthropic_beta_survives_provider_filter_on_passthrough_path(config): + from litellm.anthropic_beta_headers_manager import update_headers_with_filtered_beta + + headers, _ = config.validate_anthropic_messages_environment( + headers={"Anthropic-Beta": "caller-flag"}, + model="some-model", + messages=[], + optional_params={"speed": "fast"}, + litellm_params={}, + api_key="sk-test", + api_base="https://host/v1", + ) + + # The deployment routes as provider "openai", which has no beta mapping, so an + # unconditional filter would drop every anthropic-beta value. The handler must + # skip filtering for this config so the native upstream still receives them. + if config.should_filter_anthropic_beta_headers(): + headers = update_headers_with_filtered_beta(headers=dict(headers), provider="openai") + + survived = set(headers.get("anthropic-beta", "").split(",")) + assert {"caller-flag", "fast-mode-2026-02-01"} <= survived + + stripped = update_headers_with_filtered_beta(headers=dict(headers), provider="openai") + assert "anthropic-beta" not in stripped