diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index ebd514c2605..7bfe49efe59 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -69,11 +69,12 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): "speed", "output_config", "reasoning_effort", - # TODO: Add Anthropic `metadata` support - # "metadata", + "metadata", ] - def _remove_scope_from_cache_control(self, anthropic_messages_request: dict) -> None: + def _remove_scope_from_cache_control( + self, anthropic_messages_request: dict + ) -> None: """ Remove `scope` field from cache_control blocks. @@ -136,7 +137,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): text = content_block.get("text", "") content_type = content_block.get("type", "") # Skip text blocks that start with billing header - if content_type == "text" and text.startswith("x-anthropic-billing-header:"): + if content_type == "text" and text.startswith( + "x-anthropic-billing-header:" + ): continue filtered_list.append(content_block) else: @@ -160,6 +163,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): def _is_system_role_message(message: Any) -> bool: return isinstance(message, dict) and message.get("role") == "system" + _CONVERTED_SYSTEM_NOTE: Final = ( "Operator note (not from the user): the following was originally a mid-conversation system-role reminder." ) @@ -215,6 +219,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): def _normalize_system_role_messages(self, anthropic_messages_request: dict, model: str) -> None: """Normalize ``role: "system"`` entries in ``messages`` per the Anthropic + ``/v1/messages`` contract, which the first-party API, Bedrock Invoke, Vertex, and Azure Foundry all enforce identically. @@ -248,6 +253,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): messages: Final = anthropic_messages_request.get("messages") if not isinstance(messages, list): return + leading_count: Final = next( (i for i, m in enumerate(messages) if not self._is_system_role_message(m)), len(messages), @@ -275,7 +281,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): ) for block in self._as_system_content_blocks(source) ] - filtered_system: Final = self._filter_billing_headers_from_system(system_content) + filtered_system: Final = self._filter_billing_headers_from_system( + system_content + ) if filtered_system: anthropic_messages_request["system"] = filtered_system else: @@ -290,7 +298,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): litellm_params: dict, stream: bool | None = None, ) -> str: - api_base = AnthropicModelInfo.get_api_base(api_base) or "https://api.anthropic.com" + api_base = ( + AnthropicModelInfo.get_api_base(api_base) or "https://api.anthropic.com" + ) if not api_base.endswith("/v1/messages"): api_base = f"{api_base}/v1/messages" return api_base @@ -306,7 +316,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): api_base: str | None = None, ) -> tuple[dict, str | None]: # Check for Anthropic OAuth token in Authorization header - headers, api_key = optionally_handle_anthropic_oauth(headers=headers, api_key=api_key) + headers, api_key = optionally_handle_anthropic_oauth( + headers=headers, api_key=api_key + ) header_names: Final = frozenset(name.lower() for name in headers) if "x-api-key" not in header_names and "authorization" not in header_names: @@ -335,7 +347,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): return headers, api_base @staticmethod - def _translate_reasoning_effort_to_anthropic(model: str, optional_params: dict, custom_llm_provider: str) -> None: + def _translate_reasoning_effort_to_anthropic( + model: str, optional_params: dict, custom_llm_provider: str + ) -> None: """Map OpenAI-style ``reasoning_effort`` to native Anthropic params. Caller-supplied ``thinking`` / ``output_config`` win over the alias. @@ -367,7 +381,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): optional_params.setdefault("thinking", mapped_thinking) if AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider): - mapped_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort) + mapped_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( + reasoning_effort + ) if mapped_effort is None: raise AnthropicError( message=( @@ -377,7 +393,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): ), status_code=400, ) - gate_error: Final = AnthropicConfig._validate_effort_for_model(model, mapped_effort, custom_llm_provider) + gate_error: Final = AnthropicConfig._validate_effort_for_model( + model, mapped_effort, custom_llm_provider + ) if gate_error is not None: raise AnthropicError(message=gate_error, status_code=400) existing_output_config = optional_params.get("output_config") @@ -399,7 +417,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): """ from litellm.llms.anthropic.chat.transformation import AnthropicConfig - if not AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider): + if not AnthropicModelInfo._is_adaptive_thinking_model( + model, custom_llm_provider + ): return if AnthropicModelInfo._supports_legacy_thinking(model, custom_llm_provider): return @@ -428,7 +448,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): @staticmethod def _translate_adaptive_effort_for_non_adaptive_model( - model: str, optional_params: dict, max_tokens: int | None, custom_llm_provider: str + model: str, + optional_params: dict, + max_tokens: int | None, + custom_llm_provider: str, ) -> None: """Translate the 4.6+ adaptive-thinking interface (``thinking.type=adaptive`` and/or ``output_config.effort``) down to what an older Anthropic model @@ -476,8 +499,12 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): output_config: Final = optional_params.get("output_config") thinking: Final = optional_params.get("thinking") - effort: Final = output_config.get("effort") if isinstance(output_config, dict) else None - adaptive_thinking: Final = isinstance(thinking, dict) and thinking.get("type") == "adaptive" + effort: Final = ( + output_config.get("effort") if isinstance(output_config, dict) else None + ) + adaptive_thinking: Final = ( + isinstance(thinking, dict) and thinking.get("type") == "adaptive" + ) if effort is None and not adaptive_thinking: return @@ -486,9 +513,14 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): # reject. Effort-only requests pass through so provider subclasses (bedrock/vertex) keep # owning level clamping; an adaptive request only stays here when its effort level is one # the model supports, otherwise it falls through to the legacy budget translation below. - if AnthropicConfig._model_supports_effort_param(model, custom_llm_provider) and ( + if AnthropicConfig._model_supports_effort_param( + model, custom_llm_provider + ) and ( not adaptive_thinking - or AnthropicConfig._validate_effort_for_model(model, effort, custom_llm_provider) is None + or AnthropicConfig._validate_effort_for_model( + model, effort, custom_llm_provider + ) + is None ): if adaptive_thinking: optional_params.pop("thinking", None) @@ -510,7 +542,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): except _BadRequestError as e: raise AnthropicError(message=str(e.message), status_code=400) capped_thinking: Final = ( - AnthropicConfig._cap_thinking_budget_to_max_tokens(legacy_thinking, max_tokens) + AnthropicConfig._cap_thinking_budget_to_max_tokens( + legacy_thinking, max_tokens + ) if legacy_thinking is not None else None ) @@ -553,8 +587,12 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): return thinking: Final = optional_params.get("thinking") output_config: Final = optional_params.get("output_config") - thinking_enabled: Final = isinstance(thinking, dict) and thinking.get("type") == "enabled" - effort_enabled: Final = isinstance(output_config, dict) and output_config.get("effort") is not None + thinking_enabled: Final = ( + isinstance(thinking, dict) and thinking.get("type") == "enabled" + ) + effort_enabled: Final = ( + isinstance(output_config, dict) and output_config.get("effort") is not None + ) if thinking_enabled or effort_enabled: optional_params.pop("temperature", None) @@ -572,7 +610,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): This takes in a request in the Anthropic /v1/messages API spec -> transforms it to /v1/messages API spec (i.e) no transformation is needed """ - max_tokens: Final = anthropic_messages_optional_request_params.pop("max_tokens", None) + max_tokens: Final = anthropic_messages_optional_request_params.pop( + "max_tokens", None + ) if max_tokens is None: raise AnthropicError( message="max_tokens is required for Anthropic /v1/messages API", @@ -612,22 +652,30 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): system_param: Final = anthropic_messages_optional_request_params.get("system") if self.should_strip_billing_metadata() and system_param is not None: - filtered_system: Final = self._filter_billing_headers_from_system(system_param) + filtered_system: Final = self._filter_billing_headers_from_system( + system_param + ) if filtered_system is not None and len(filtered_system) > 0: anthropic_messages_optional_request_params["system"] = filtered_system else: anthropic_messages_optional_request_params.pop("system", None) # Transform context_management from OpenAI format to Anthropic format if needed - context_management_param: Final = anthropic_messages_optional_request_params.get("context_management") + context_management_param: Final = ( + anthropic_messages_optional_request_params.get("context_management") + ) if context_management_param is not None: from litellm.llms.anthropic.chat.transformation import AnthropicConfig - transformed_context_management: Final = AnthropicConfig.map_openai_context_management_to_anthropic( - context_management_param + transformed_context_management: Final = ( + AnthropicConfig.map_openai_context_management_to_anthropic( + context_management_param + ) ) if transformed_context_management is not None: - anthropic_messages_optional_request_params["context_management"] = transformed_context_management + anthropic_messages_optional_request_params["context_management"] = ( + transformed_context_management + ) ####### get required params for all anthropic messages requests ###### # Lazy %s: the f-string previously stringified the entire messages @@ -638,15 +686,20 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): # Auto-strip advisor blocks from history if advisor tool is absent. # Prevents Anthropic 400: advisor_tool_result in history requires advisor tool. _tools: Final = anthropic_messages_optional_request_params.get("tools") or [] - _has_advisor: Final = any(isinstance(t, dict) and t.get("type") == ANTHROPIC_ADVISOR_TOOL_TYPE for t in _tools) + _has_advisor: Final = any( + isinstance(t, dict) and t.get("type") == ANTHROPIC_ADVISOR_TOOL_TYPE + for t in _tools + ) if not _has_advisor: messages = strip_advisor_blocks_from_messages(messages) - anthropic_messages_request: Final[AnthropicMessagesRequest] = AnthropicMessagesRequest( - messages=messages, - max_tokens=max_tokens, - model=model, - **anthropic_messages_optional_request_params, + anthropic_messages_request: Final[AnthropicMessagesRequest] = ( + AnthropicMessagesRequest( + messages=messages, + max_tokens=max_tokens, + model=model, + **anthropic_messages_optional_request_params, + ) ) return dict(anthropic_messages_request) @@ -661,8 +714,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): """ try: raw_response_json: Final = raw_response.json() - except Exception: - raise AnthropicError(message=raw_response.text, status_code=raw_response.status_code) + except Exception: # noqa: BLE001 + raise AnthropicError( + message=raw_response.text, status_code=raw_response.status_code + ) return AnthropicMessagesResponse(**raw_response_json) def get_async_streaming_response_iterator( @@ -736,7 +791,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): # Add context management header if any other edits exist if has_other: - beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value) + beta_values.add( + ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value + ) # Check for structured outputs. Anthropic's newer request shape nests # the schema under output_config.format; the older top-level @@ -745,7 +802,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): if optional_params.get("output_format") is not None or ( isinstance(output_config, dict) and output_config.get("format") is not None ): - beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.STRUCTURED_OUTPUT_2025_09_25.value) + beta_values.add( + ANTHROPIC_BETA_HEADER_VALUES.STRUCTURED_OUTPUT_2025_09_25.value + ) # Check for fast mode if optional_params.get("speed") == "fast": @@ -755,8 +814,13 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): tools = optional_params.get("tools") if tools: for tool in tools: - if isinstance(tool, dict) and tool.get("type") == ANTHROPIC_ADVISOR_TOOL_TYPE: - beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.ADVISOR_TOOL_2026_03_01.value) + if ( + isinstance(tool, dict) + and tool.get("type") == ANTHROPIC_ADVISOR_TOOL_TYPE + ): + beta_values.add( + ANTHROPIC_BETA_HEADER_VALUES.ADVISOR_TOOL_2026_03_01.value + ) break # Check for tool search tools @@ -765,7 +829,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): anthropic_model_info: Final = AnthropicModelInfo() if anthropic_model_info.is_tool_search_used(tools): # Use provider-specific tool search header - tool_search_header: Final = get_tool_search_beta_header(custom_llm_provider) + tool_search_header: Final = get_tool_search_beta_header( + custom_llm_provider + ) beta_values.add(tool_search_header) if beta_values: diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_messages_metadata.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_messages_metadata.py new file mode 100644 index 00000000000..9f0c11d87fb --- /dev/null +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_messages_metadata.py @@ -0,0 +1,72 @@ +""" +Tests for Anthropic Messages passthrough metadata support. + +Related issue: https://github.com/BerriAI/litellm/issues/30663 +""" + +from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( + AnthropicMessagesConfig, +) +from litellm.types.router import GenericLiteLLMParams + + +class TestAnthropicMessagesMetadataSupport: + """Test that metadata is properly supported in Anthropic Messages passthrough.""" + + def setup_method(self): + self.config = AnthropicMessagesConfig() + + def test_metadata_in_supported_params(self): + """Verify 'metadata' is listed in supported Anthropic Messages params.""" + supported_params = self.config.get_supported_anthropic_messages_params( + model="claude-sonnet-4-20250514" + ) + assert ( + "metadata" in supported_params + ), "'metadata' should be in supported params for Anthropic Messages passthrough" + + def test_metadata_appears_exactly_once(self): + """Verify 'metadata' is not duplicated in the supported params list.""" + supported_params = self.config.get_supported_anthropic_messages_params( + model="claude-sonnet-4-20250514" + ) + assert supported_params.count("metadata") == 1 + + def test_core_params_still_present(self): + """Regression: ensure adding metadata did not remove existing params.""" + supported_params = self.config.get_supported_anthropic_messages_params( + model="claude-sonnet-4-20250514" + ) + expected_core_params = [ + "messages", + "model", + "system", + "max_tokens", + "temperature", + ] + for param in expected_core_params: + assert param in supported_params + + def test_metadata_forwarded_in_transformed_request(self): + """Verify 'metadata' is actually forwarded in the final transformed Anthropic Messages request body.""" + result = self.config.transform_anthropic_messages_request( + model="claude-sonnet-4-20250514", + messages=[ + { + "role": "user", + "content": "Hello", + } + ], + anthropic_messages_optional_request_params={ + "max_tokens": 10, + "metadata": { + "user_id": "test-user-123", + }, + }, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert result["metadata"] == { + "user_id": "test-user-123", + } \ No newline at end of file