diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 9d1e921cce4..8bf5478901e 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -7,8 +7,9 @@ import asyncio import contextvars -from collections.abc import AsyncIterator, Coroutine, Iterator +from collections.abc import AsyncIterator, Coroutine, Iterator, Mapping, Sequence from functools import partial +from types import MappingProxyType from typing import Any, Final, cast import litellm @@ -418,6 +419,32 @@ def validate_anthropic_api_metadata(metadata: dict | None = None) -> dict | None return anthropic_metadata_obj.model_dump(exclude_none=True) +def _drop_unsupported_anthropic_messages_params( + anthropic_messages_optional_request_params: Mapping[str, Any], + model: str, + custom_llm_provider: str | None, + additional_drop_params: Sequence[str] | None = None, +) -> Mapping[str, Any]: + from litellm.llms.anthropic.chat.transformation import AnthropicConfig + + additional_drop_params_list: Final = additional_drop_params if isinstance(additional_drop_params, list) else None + supports_effort: Final = AnthropicConfig._model_supports_effort_param(model, custom_llm_provider or "anthropic") # pyright: ignore[reportPrivateUsage] # shared Anthropic capability gate + supports_reasoning: Final = AnthropicConfig._supports_model_capability( # pyright: ignore[reportPrivateUsage] # shared Anthropic capability gate + model, "supports_reasoning", custom_llm_provider or "anthropic" + ) + drop_context_management: Final = custom_llm_provider in ("vertex_ai", "bedrock") and "haiku" in model.lower() + return MappingProxyType( + { + k: v + for k, v in anthropic_messages_optional_request_params.items() + if not (additional_drop_params_list is not None and k in additional_drop_params_list) + and (k != "output_config" or supports_effort) + and (k != "thinking" or supports_reasoning) + and (k != "context_management" or not drop_context_management) + } + ) + + def anthropic_messages_handler( max_tokens: int, messages: list[dict], @@ -641,19 +668,47 @@ def anthropic_messages_handler( custom_llm_provider=custom_llm_provider, ) ) - if is_reasoning_auto_summary_enabled(): - thinking_param: Final = anthropic_messages_optional_request_params.get("thinking") - if isinstance(thinking_param, dict) and thinking_param.get("type") != "disabled": - anthropic_messages_optional_request_params["thinking"] = { - **thinking_param, - "display": "summarized", + + should_drop_params: Final = ( + litellm.drop_params is True + or getattr(litellm_params, "drop_params", None) is True + or kwargs.get("drop_params") is True + ) + + filtered_anthropic_messages_params: Final = ( + _drop_unsupported_anthropic_messages_params( + anthropic_messages_optional_request_params=anthropic_messages_optional_request_params, + model=model, + custom_llm_provider=custom_llm_provider, + additional_drop_params=kwargs.get("additional_drop_params"), + ) + if should_drop_params + else anthropic_messages_optional_request_params + ) + thinking_param: Final = filtered_anthropic_messages_params.get("thinking") + final_anthropic_messages_params: Final = ( + MappingProxyType( + { + **filtered_anthropic_messages_params, + "thinking": { # mutable-ok: construct the summarized payload before freezing + **thinking_param, + "display": "summarized", + }, } + ) + if is_reasoning_auto_summary_enabled() + and isinstance(thinking_param, dict) + and thinking_param.get("type") != "disabled" + else filtered_anthropic_messages_params + ) return base_llm_http_handler.anthropic_messages_handler( model=model, messages=strip_provider_specific_fields_from_anthropic_messages(messages), anthropic_messages_provider_config=anthropic_messages_provider_config, - anthropic_messages_optional_request_params=dict(anthropic_messages_optional_request_params), + anthropic_messages_optional_request_params=dict( # mutable-ok: downstream handler requires a dict + final_anthropic_messages_params + ), _is_async=is_async, client=client, custom_llm_provider=custom_llm_provider, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 01f7a2fb7ab..d55ba45debc 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -42,8 +42,8 @@ def test_anthropic_experimental_pass_through_messages_handler(): model="openai/claude-3-5-sonnet-20240620", api_key="test-api-key", ) - except (ValueError, TypeError, AttributeError) as e: - print(f"Error: {e}") + except (ValueError, TypeError, AttributeError): + pass mock_responses.assert_called_once() assert mock_responses.call_args.kwargs["api_key"] == "test-api-key" @@ -129,8 +129,8 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an api_base="test-api-base", custom_key="custom_value", ) - except (ValueError, TypeError, AttributeError) as e: - print(f"Error: {e}") + except (ValueError, TypeError, AttributeError): + pass mock_completion.assert_called_once() assert mock_completion.call_args.kwargs["api_key"] == "test-api-key" assert mock_completion.call_args.kwargs["api_base"] == "test-api-base" @@ -244,8 +244,8 @@ def test_anthropic_experimental_pass_through_messages_handler_custom_llm_provide custom_llm_provider="my-custom-llm", api_key="test-api-key", ) - except (ValueError, TypeError, AttributeError) as e: - print(f"Error: {e}") + except (ValueError, TypeError, AttributeError): + pass # Assert that litellm.completion was called when using a custom LLM provider mock_completion.assert_called_once() @@ -296,7 +296,6 @@ async def test_bedrock_converse_budget_tokens_preserved(): mock_acompletion.assert_called_once() call_kwargs = mock_acompletion.call_args.kwargs - print("acompletion call kwargs: ", json.dumps(call_kwargs, indent=4, default=str)) # Verify thinking parameter is passed through with budget_tokens preserved thinking_param = call_kwargs.get("thinking") @@ -328,8 +327,8 @@ def test_openai_model_with_thinking_converts_to_reasoning(): api_key="test-api-key", thinking={"type": "enabled", "budget_tokens": 1024}, ) - except (ValueError, TypeError, AttributeError) as e: - print(f"Error: {e}") + except (ValueError, TypeError, AttributeError): + pass mock_responses.assert_called_once() @@ -1438,3 +1437,93 @@ async def test_anthropic_messages_leaves_non_provider_failures_unmapped(): ) assert "Traceback" not in str(excinfo.value) + + +@pytest.mark.asyncio +async def test_anthropic_pass_through_drop_params(monkeypatch): + from litellm.llms.anthropic.experimental_pass_through.messages import handler + + mock_handler = AsyncMock() + monkeypatch.setattr(handler.base_llm_http_handler, "anthropic_messages_handler", mock_handler) + + await handler.anthropic_messages_handler( + model="vertex_ai/claude-3-haiku-20240307", + messages=[{"role": "user", "content": "hello"}], + max_tokens=64, + drop_params=True, + context_management={"edits": []}, + custom_llm_provider="vertex_ai", + is_async=True, + ) + + optional_params = mock_handler.call_args.kwargs.get("anthropic_messages_optional_request_params", {}) + assert "context_management" not in optional_params + + +def test_drop_params_filters_unsupported_anthropic_params(): + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + _drop_unsupported_anthropic_messages_params, + ) + + params = { + "max_tokens": 64, + "metadata": {"request_id": "test"}, + "thinking": {"type": "enabled", "budget_tokens": 1024}, + "output_config": {"effort": "high"}, + "context_management": {"edits": []}, + } + + filtered = _drop_unsupported_anthropic_messages_params( + anthropic_messages_optional_request_params=params, + model="claude-3-haiku-20240307", + custom_llm_provider="vertex_ai", + additional_drop_params=["metadata"], + ) + + assert filtered == {"max_tokens": 64} + assert params["thinking"] == {"type": "enabled", "budget_tokens": 1024} + assert "context_management" in params + + +def test_drop_params_preserves_supported_anthropic_params(): + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + _drop_unsupported_anthropic_messages_params, + ) + + params = { + "thinking": {"type": "enabled", "budget_tokens": 1024}, + "output_config": {"effort": "high"}, + "context_management": {"edits": []}, + } + + filtered = _drop_unsupported_anthropic_messages_params( + anthropic_messages_optional_request_params=params, + model="claude-opus-4-5-20251101", + custom_llm_provider="vertex_ai", + ) + + assert filtered == params + assert filtered is not params + + +@pytest.mark.asyncio +async def test_anthropic_pass_through_keeps_supported_params_without_drop(monkeypatch): + import litellm + from litellm.llms.anthropic.experimental_pass_through.messages import handler + + mock_handler = AsyncMock() + monkeypatch.setattr(litellm, "drop_params", False) + monkeypatch.setattr(handler.base_llm_http_handler, "anthropic_messages_handler", mock_handler) + + await handler.anthropic_messages_handler( + model="vertex_ai/claude-opus-4-5-20251101", + messages=[{"role": "user", "content": "hello"}], + max_tokens=64, + thinking={"type": "enabled", "budget_tokens": 1024}, + drop_params=False, + custom_llm_provider="vertex_ai", + is_async=True, + ) + + optional_params = mock_handler.call_args.kwargs["anthropic_messages_optional_request_params"] + assert optional_params["thinking"] == {"type": "enabled", "budget_tokens": 1024}