From 7b524233801405fbe2e92ceea8e237110018014b Mon Sep 17 00:00:00 2001 From: shrey kharbanda Date: Thu, 1 Oct 2026 16:48:54 -0700 Subject: [PATCH] fix(anthropic): add missing Azure and Vertex beta headers --- litellm/anthropic_beta_headers_config.json | 1 + litellm/llms/anthropic/common_utils.py | 10 ++- .../llms/azure_ai/anthropic/transformation.py | 5 ++ .../transformation.py | 19 ++++- .../anthropic/transformation.py | 3 + litellm/types/llms/anthropic.py | 3 + .../anthropic/test_anthropic_common_utils.py | 2 +- ...azure_anthropic_messages_transformation.py | 42 ++++++++++ .../test_azure_anthropic_transformation.py | 54 +++++++++++++ ...artner_models_anthropic_messages_config.py | 76 +++++++++++++++++++ ...partner_models_anthropic_transformation.py | 48 ++++++++++++ 11 files changed, 257 insertions(+), 6 deletions(-) diff --git a/litellm/anthropic_beta_headers_config.json b/litellm/anthropic_beta_headers_config.json index 332d9b3ad9d..8e2b7b553ab 100644 --- a/litellm/anthropic_beta_headers_config.json +++ b/litellm/anthropic_beta_headers_config.json @@ -201,6 +201,7 @@ "mcp-client-2025-11-20": null, "mcp-client-2025-04-04": null, "mcp-servers-2025-12-04": null, + "mid-conversation-tool-changes-2026-07-01": "mid-conversation-tool-changes-2026-07-01", "output-128k-2025-02-19": null, "structured-output-2024-03-01": null, "per-turn-control-2026-07-01": "per-turn-control-2026-07-01", diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index 65c2fccceeb..3226898e850 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -37,10 +37,10 @@ from litellm.types.llms.anthropic import ( ANTHROPIC_OAUTH_BETA_HEADER, ANTHROPIC_OAUTH_TOKEN_PREFIX, ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER, + LEGACY_THINKING_DISPLAY_UPDATES_PROVIDERS, AllAnthropicToolsValues, AnthropicMcpServerTool, AnthropicMessagesToolChoice, - AnthropicThinkingParam, ) from litellm.types.llms.openai import AllMessageValues from litellm.types.proxy.model_listing import ModelInfoResponse @@ -341,7 +341,7 @@ class AnthropicModelInfo(BaseLLMModelInfo): file_ids: Final = get_file_ids_from_messages(messages) return len(file_ids) > 0 - def is_thinking_display_updates_used(self, thinking: AnthropicThinkingParam | None) -> bool: + def is_thinking_display_updates_used(self, thinking: object) -> bool: if not isinstance(thinking, dict): return False return thinking.get("type") in ("adaptive", "enabled") and thinking.get("display") == "updates" @@ -771,7 +771,11 @@ class AnthropicModelInfo(BaseLLMModelInfo): ) existing_output_config: Final = optional_params.get("output_config") display: Final = thinking.get("display") - if display in ("summarized", "omitted"): + preserve_display: Final = display in ("summarized", "omitted") + preserve_updates: Final = ( + display == "updates" and custom_llm_provider in LEGACY_THINKING_DISPLAY_UPDATES_PROVIDERS + ) + if preserve_display or preserve_updates: optional_params["thinking"] = {"type": "adaptive", "display": display} else: optional_params["thinking"] = {"type": "adaptive"} diff --git a/litellm/llms/azure_ai/anthropic/transformation.py b/litellm/llms/azure_ai/anthropic/transformation.py index 864d2134a84..41524e240f0 100644 --- a/litellm/llms/azure_ai/anthropic/transformation.py +++ b/litellm/llms/azure_ai/anthropic/transformation.py @@ -96,9 +96,14 @@ class AzureAnthropicConfig(AnthropicConfig): is_vertex_request=optional_params.get("is_vertex_request", False), user_anthropic_beta_headers=user_anthropic_beta_headers, mcp_server_used=mcp_server_used, + is_thinking_display_updates_used=self.is_thinking_display_updates_used( + optional_params.get("thinking"), # pyright: ignore[reportUnknownArgumentType] # detector accepts unvalidated input + ), ) # Merge headers - Azure auth (api-key or Authorization) takes precedence headers = {**anthropic_headers, **headers} + if "anthropic-beta" in anthropic_headers: + headers["anthropic-beta"] = anthropic_headers["anthropic-beta"] # Ensure anthropic-version header is set if "anthropic-version" not in headers: diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py index be2dacd23c2..98646c125a9 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py @@ -8,6 +8,8 @@ from litellm.llms.anthropic.pass_through.messages.transformation import ( from litellm.types.llms.anthropic import ( ANTHROPIC_BETA_HEADER_VALUES, ANTHROPIC_HOSTED_TOOLS, + ANTHROPIC_MID_CONVERSATION_TOOL_CHANGES_BETA_HEADER, + ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER, ) from litellm.types.llms.anthropic_tool_search import get_tool_search_beta_header from litellm.types.llms.vertex_ai import VertexPartnerProvider @@ -115,8 +117,21 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert if _messages_carry_output_config(messages): beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.PER_TURN_CONTROL_2026_07_01.value) - if beta_values: - headers["anthropic-beta"] = ",".join(beta_values) + thinking_display_betas: Final = ( + (ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,) + if anthropic_model_info.is_thinking_display_updates_used( + optional_params.get("thinking"), # pyright: ignore[reportUnknownArgumentType] # detector accepts unvalidated input + ) + else () + ) + tool_change_betas: Final = ( + (ANTHROPIC_MID_CONVERSATION_TOOL_CHANGES_BETA_HEADER,) + if anthropic_model_info.is_mid_conversation_tool_change_used(messages) + else () + ) + all_beta_values: Final = beta_values.union(thinking_display_betas, tool_change_betas) + if all_beta_values: + headers["anthropic-beta"] = ",".join(sorted(all_beta_values)) return headers, api_base diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index 2fab6f438f6..8b581ac681d 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -120,6 +120,9 @@ class VertexAIAnthropicConfig(AnthropicConfig): file_id_used=self.is_file_id_used(messages), mcp_server_used=self.is_mcp_server_used(optional_params.get("mcp_servers")), custom_llm_provider="vertex_ai", + is_thinking_display_updates_used=self.is_thinking_display_updates_used( + data.get("thinking"), # pyright: ignore[reportUnknownArgumentType] # detector accepts unvalidated input + ), ) beta_set: Final = set(auto_betas) diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py index ee357cd6581..5ab2a1abaad 100644 --- a/litellm/types/llms/anthropic.py +++ b/litellm/types/llms/anthropic.py @@ -736,6 +736,9 @@ class AnthropicThinkingParam(TypedDict, total=False): display: ReadOnly[Literal["summarized", "omitted", "updates"]] +LEGACY_THINKING_DISPLAY_UPDATES_PROVIDERS: Final = frozenset(("azure_ai", "vertex_ai")) + + class ANTHROPIC_HOSTED_TOOLS(str, Enum): WEB_SEARCH = "web_search" BASH = "bash" diff --git a/tests/unit/llms/anthropic/test_anthropic_common_utils.py b/tests/unit/llms/anthropic/test_anthropic_common_utils.py index 68e2e9650d5..2235c2cd867 100644 --- a/tests/unit/llms/anthropic/test_anthropic_common_utils.py +++ b/tests/unit/llms/anthropic/test_anthropic_common_utils.py @@ -2431,7 +2431,7 @@ def test_thinking_display_beta_requires_active_thinking(thinking: object, expect ( ("summarized", {"type": "adaptive", "display": "summarized"}), ("omitted", {"type": "adaptive", "display": "omitted"}), - ("updates", {"type": "adaptive"}), + ("updates", {"type": "adaptive", "display": "updates"}), ), ) def test_shared_legacy_thinking_translation_preserves_supported_display( diff --git a/tests/unit/llms/azure_ai/claude/test_azure_anthropic_messages_transformation.py b/tests/unit/llms/azure_ai/claude/test_azure_anthropic_messages_transformation.py index b78b2d0d842..02b29a0a260 100644 --- a/tests/unit/llms/azure_ai/claude/test_azure_anthropic_messages_transformation.py +++ b/tests/unit/llms/azure_ai/claude/test_azure_anthropic_messages_transformation.py @@ -470,3 +470,45 @@ def test_azure_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_fl and info.get("supports_mid_conversation_system") is not True ] assert missing == [] + + +@pytest.mark.usefixtures("local_model_cost_map") +@pytest.mark.parametrize("feature", ("thinking", "output_config", "tool_addition", "tool_removal", "none")) +@pytest.mark.parametrize("explicit_beta", (False, True)) +def test_messages_feature_beta_headers(feature: str, explicit_beta: bool) -> None: + from typing import Final + + from litellm.llms.azure_ai.anthropic.messages_transformation import AzureAnthropicMessagesConfig + from litellm.types.llms.anthropic import ( + ANTHROPIC_BETA_HEADER_VALUES, + ANTHROPIC_MID_CONVERSATION_TOOL_CHANGES_BETA_HEADER, + ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER, + ) + + beta: Final = ( + ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER + if feature == "thinking" + else ANTHROPIC_BETA_HEADER_VALUES.PER_TURN_CONTROL_2026_07_01.value + if feature == "output_config" + else ANTHROPIC_MID_CONVERSATION_TOOL_CHANGES_BETA_HEADER + ) + message: Final = ( + {"role": "system", "content": [], "output_config": {"effort": "high"}} + if feature == "output_config" + else {"role": "system", "content": [{"type": feature, "tool": {"type": "tool_reference", "name": "ping"}}]} + if feature in ("tool_addition", "tool_removal") + else {"role": "user", "content": "Reply with OK"} + ) + headers, _ = AzureAnthropicMessagesConfig().validate_anthropic_messages_environment( + headers={"anthropic-beta": f"existing-beta,{beta}" if explicit_beta else "existing-beta"}, + model="claude-fable-5-1", + messages=[{"role": "user", "content": "Hello"}, message], + optional_params={"thinking": {"type": "adaptive", "display": "updates"}} if feature == "thinking" else {}, + litellm_params={"vertex_ai_project": "test-project", "vertex_ai_location": "us-east5"}, + api_key="test-key", + api_base="https://example.com/anthropic", + ) + + betas: Final = headers["anthropic-beta"].split(",") + assert betas.count(beta) == int(feature != "none" or explicit_beta) + assert betas.count("existing-beta") == 1 diff --git a/tests/unit/llms/azure_ai/claude/test_azure_anthropic_transformation.py b/tests/unit/llms/azure_ai/claude/test_azure_anthropic_transformation.py index ddbc168589a..0cc887b6c98 100644 --- a/tests/unit/llms/azure_ai/claude/test_azure_anthropic_transformation.py +++ b/tests/unit/llms/azure_ai/claude/test_azure_anthropic_transformation.py @@ -485,3 +485,57 @@ def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_mo "content": [{"type": "text", "text": "Answer with exactly one word."}], } + + +@pytest.mark.usefixtures("local_model_cost_map") +@pytest.mark.parametrize("thinking_type", ("adaptive", "enabled")) +@pytest.mark.parametrize("display", ("summarized", "omitted", "updates")) +def test_chat_preserves_thinking_display_and_beta(thinking_type: str, display: str) -> None: + from typing import Final + + from litellm.llms.azure_ai.anthropic.transformation import AzureAnthropicConfig + from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER + + config: Final = AzureAnthropicConfig() + optional_params: Final = config.map_openai_params( + non_default_params={"thinking": {"type": thinking_type, "display": display, "budget_tokens": 2048}}, + optional_params={}, + model="claude-fable-5-1", + drop_params=False, + ) + headers: Final = config.validate_environment( + headers={"anthropic-beta": "existing-beta"}, + model="claude-fable-5-1", + messages=[{"role": "user", "content": "Reply with OK"}], + optional_params=optional_params, + litellm_params={}, + api_key="test-key", + ) + result: Final = config.transform_request( + model="claude-fable-5-1", + messages=[{"role": "user", "content": "Reply with OK"}], + optional_params=optional_params, + litellm_params={}, + headers=headers, + ) + + assert result["thinking"]["type"] == "adaptive" + assert result["thinking"]["display"] == display + assert (ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in headers["anthropic-beta"].split(",")) == (display == "updates") + assert "existing-beta" in headers["anthropic-beta"].split(",") + + +@pytest.mark.parametrize("thinking", ("enabled", {"type": "bogus"}, {"type": "adaptive", "display": "future-mode"})) +def test_beta_detection_preserves_unrecognized_thinking(thinking: object) -> None: + from typing import Final + + from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER + + params: Final = {"thinking": thinking} + headers: Final = AzureAnthropicConfig().validate_environment( + headers={}, model="claude-fable-5-1", messages=[], optional_params=params, + litellm_params={}, api_key="test-key", + ) + + assert params["thinking"] == thinking + assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER not in headers.get("anthropic-beta", "").split(",") diff --git a/tests/unit/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py b/tests/unit/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py index 67a32cc82bc..316d785ebc9 100644 --- a/tests/unit/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py +++ b/tests/unit/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py @@ -726,3 +726,79 @@ def test_vertex_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_f and info.get("supports_mid_conversation_system") is not True ] assert missing == [] + + +@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config") +@pytest.mark.parametrize("feature", ("thinking", "output_config", "tool_addition", "tool_removal", "none")) +@pytest.mark.parametrize("explicit_beta", (False, True)) +def test_messages_feature_beta_headers(feature: str, explicit_beta: bool) -> None: + from typing import Final + from unittest.mock import patch + + from google.oauth2.credentials import Credentials + from litellm.anthropic_beta_headers_manager import update_headers_with_filtered_beta + from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.experimental_pass_through.transformation import VertexAIPartnerModelsAnthropicMessagesConfig + from litellm.types.llms.anthropic import ( + ANTHROPIC_BETA_HEADER_VALUES, + ANTHROPIC_MID_CONVERSATION_TOOL_CHANGES_BETA_HEADER, + ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER, + ) + + beta: Final = ( + ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER + if feature == "thinking" + else ANTHROPIC_BETA_HEADER_VALUES.PER_TURN_CONTROL_2026_07_01.value + if feature == "output_config" + else ANTHROPIC_MID_CONVERSATION_TOOL_CHANGES_BETA_HEADER + ) + message: Final = ( + {"role": "system", "content": [], "output_config": {"effort": "high"}} + if feature == "output_config" + else {"role": "system", "content": [{"type": feature, "tool": {"type": "tool_reference", "name": "ping"}}]} + if feature in ("tool_addition", "tool_removal") + else {"role": "user", "content": "Reply with OK"} + ) + with patch( + "google.auth.default", + return_value=(MagicMock(spec=Credentials, token="test-token", expired=False), "test-project"), + ): + headers, _ = VertexAIPartnerModelsAnthropicMessagesConfig().validate_anthropic_messages_environment( + headers={"anthropic-beta": f"existing-beta,{beta}" if explicit_beta else "existing-beta"}, + model="claude-fable-5-1", + messages=[{"role": "user", "content": "Hello"}, message], + optional_params={"thinking": {"type": "adaptive", "display": "updates"}} if feature == "thinking" else {}, + litellm_params={"vertex_ai_project": "test-project", "vertex_ai_location": "us-east5"}, + api_key="test-key", + api_base="https://example.com/anthropic", + ) + + betas: Final = headers["anthropic-beta"].split(",") + assert betas.count(beta) == int(feature != "none" or explicit_beta) + assert betas.count("existing-beta") == 1 + + if feature in ("tool_addition", "tool_removal"): + filtered_headers: Final = update_headers_with_filtered_beta(headers=headers, provider="vertex_ai") + assert filtered_headers.get("anthropic-beta", "").split(",").count(beta) == 1 + + +@pytest.mark.parametrize("thinking", ("enabled", {"type": "bogus"}, {"type": "adaptive", "display": "future-mode"})) +def test_beta_detection_preserves_unrecognized_thinking(thinking: object) -> None: + from typing import Final + from unittest.mock import patch + + from google.oauth2.credentials import Credentials + from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER + + params: Final = {"thinking": thinking} + with patch( + "google.auth.default", + return_value=(MagicMock(spec=Credentials, token="test-token", expired=False), "test-project"), + ): + headers, _ = VertexAIPartnerModelsAnthropicMessagesConfig().validate_anthropic_messages_environment( + headers={}, model="claude-fable-5-1", messages=[], optional_params=params, + litellm_params={"vertex_ai_project": "test-project", "vertex_ai_location": "us-east5"}, + api_key="test-key", api_base="https://example.com/anthropic", + ) + + assert params["thinking"] == thinking + assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER not in headers.get("anthropic-beta", "").split(",") diff --git a/tests/unit/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/unit/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index ca4a3dedb4a..dc1886e171f 100644 --- a/tests/unit/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/unit/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -875,3 +875,51 @@ def test_chat_flagged_model_replays_a_byte_identical_prefix_around_a_mid_convers _assert_prefix_stable(requests) assert [m["role"] for m in requests[1]["messages"]] == ["user", "assistant", "user", "system"] assert [m["role"] for m in requests[2]["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"] + + +@pytest.mark.usefixtures("local_model_cost_map") +@pytest.mark.parametrize("thinking_type", ("adaptive", "enabled")) +@pytest.mark.parametrize("display", ("summarized", "omitted", "updates")) +def test_chat_preserves_thinking_display_and_beta(thinking_type: str, display: str) -> None: + from typing import Final + + from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import VertexAIAnthropicConfig + from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER + + config: Final = VertexAIAnthropicConfig() + optional_params: Final = config.map_openai_params( + non_default_params={"thinking": {"type": thinking_type, "display": display, "budget_tokens": 2048}}, + optional_params={}, + model="claude-fable-5-1", + drop_params=False, + ) + optional_params["extra_headers"] = {"anthropic-beta": "existing-beta"} + headers: Final = {} + result: Final = config.transform_request( + model="claude-fable-5-1", + messages=[{"role": "user", "content": "Reply with OK"}], + optional_params=optional_params, + litellm_params={}, + headers=headers, + ) + + assert result["thinking"]["type"] == "adaptive" + assert result["thinking"]["display"] == display + assert (ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in headers["anthropic-beta"].split(",")) == (display == "updates") + assert "existing-beta" in headers["anthropic-beta"].split(",") + + +@pytest.mark.parametrize("thinking", ("enabled", {"type": "bogus"}, {"type": "adaptive", "display": "future-mode"})) +def test_beta_detection_preserves_unrecognized_thinking(thinking: object) -> None: + from typing import Final + + from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER + + headers: Final = {} + result: Final = VertexAIAnthropicConfig().transform_request( + model="claude-fable-5-1", messages=[{"role": "user", "content": "Hello"}], + optional_params={"thinking": thinking}, litellm_params={}, headers=headers, + ) + + assert result["thinking"] == thinking + assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER not in headers.get("anthropic-beta", "").split(",")