This commit is contained in:
Hasnaat hussain 2026-09-08 18:50:17 +00:00 • committed by GitHub
commit 69a740d228
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 161 additions and 17 deletions

View file

@ -7,8 +7,9 @@
import asyncio
import contextvars
from collections.abc import AsyncIterator, Coroutine, Iterator
from collections.abc import AsyncIterator, Coroutine, Iterator, Mapping, Sequence
from functools import partial
from types import MappingProxyType
from typing import Any, Final, cast
import litellm
@ -418,6 +419,32 @@ def validate_anthropic_api_metadata(metadata: dict | None = None) -> dict | None
return anthropic_metadata_obj.model_dump(exclude_none=True)
def _drop_unsupported_anthropic_messages_params(
anthropic_messages_optional_request_params: Mapping[str, Any],
model: str,
custom_llm_provider: str | None,
additional_drop_params: Sequence[str] | None = None,
) -> Mapping[str, Any]:
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
additional_drop_params_list: Final = additional_drop_params if isinstance(additional_drop_params, list) else None
supports_effort: Final = AnthropicConfig._model_supports_effort_param(model, custom_llm_provider or "anthropic") # pyright: ignore[reportPrivateUsage] # shared Anthropic capability gate
supports_reasoning: Final = AnthropicConfig._supports_model_capability( # pyright: ignore[reportPrivateUsage] # shared Anthropic capability gate
model, "supports_reasoning", custom_llm_provider or "anthropic"
)
drop_context_management: Final = custom_llm_provider in ("vertex_ai", "bedrock") and "haiku" in model.lower()
return MappingProxyType(
{
k: v
for k, v in anthropic_messages_optional_request_params.items()
if not (additional_drop_params_list is not None and k in additional_drop_params_list)
and (k != "output_config" or supports_effort)
and (k != "thinking" or supports_reasoning)
and (k != "context_management" or not drop_context_management)
}
)
def anthropic_messages_handler(
max_tokens: int,
messages: list[dict],
@ -641,19 +668,47 @@ def anthropic_messages_handler(
custom_llm_provider=custom_llm_provider,
)
)
if is_reasoning_auto_summary_enabled():
thinking_param: Final = anthropic_messages_optional_request_params.get("thinking")
if isinstance(thinking_param, dict) and thinking_param.get("type") != "disabled":
anthropic_messages_optional_request_params["thinking"] = {
**thinking_param,
"display": "summarized",
should_drop_params: Final = (
litellm.drop_params is True
or getattr(litellm_params, "drop_params", None) is True
or kwargs.get("drop_params") is True
)
filtered_anthropic_messages_params: Final = (
_drop_unsupported_anthropic_messages_params(
anthropic_messages_optional_request_params=anthropic_messages_optional_request_params,
model=model,
custom_llm_provider=custom_llm_provider,
additional_drop_params=kwargs.get("additional_drop_params"),
)
if should_drop_params
else anthropic_messages_optional_request_params
)
thinking_param: Final = filtered_anthropic_messages_params.get("thinking")
final_anthropic_messages_params: Final = (
MappingProxyType(
{
**filtered_anthropic_messages_params,
"thinking": { # mutable-ok: construct the summarized payload before freezing
**thinking_param,
"display": "summarized",
},
}
)
if is_reasoning_auto_summary_enabled()
and isinstance(thinking_param, dict)
and thinking_param.get("type") != "disabled"
else filtered_anthropic_messages_params
)
return base_llm_http_handler.anthropic_messages_handler(
model=model,
messages=strip_provider_specific_fields_from_anthropic_messages(messages),
anthropic_messages_provider_config=anthropic_messages_provider_config,
anthropic_messages_optional_request_params=dict(anthropic_messages_optional_request_params),
anthropic_messages_optional_request_params=dict( # mutable-ok: downstream handler requires a dict
final_anthropic_messages_params
),
_is_async=is_async,
client=client,
custom_llm_provider=custom_llm_provider,

View file

@ -42,8 +42,8 @@ def test_anthropic_experimental_pass_through_messages_handler():
model="openai/claude-3-5-sonnet-20240620",
api_key="test-api-key",
)
except (ValueError, TypeError, AttributeError) as e:
print(f"Error: {e}")
except (ValueError, TypeError, AttributeError):
pass
mock_responses.assert_called_once()
assert mock_responses.call_args.kwargs["api_key"] == "test-api-key"
@ -129,8 +129,8 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an
api_base="test-api-base",
custom_key="custom_value",
)
except (ValueError, TypeError, AttributeError) as e:
print(f"Error: {e}")
except (ValueError, TypeError, AttributeError):
pass
mock_completion.assert_called_once()
assert mock_completion.call_args.kwargs["api_key"] == "test-api-key"
assert mock_completion.call_args.kwargs["api_base"] == "test-api-base"
@ -244,8 +244,8 @@ def test_anthropic_experimental_pass_through_messages_handler_custom_llm_provide
custom_llm_provider="my-custom-llm",
api_key="test-api-key",
)
except (ValueError, TypeError, AttributeError) as e:
print(f"Error: {e}")
except (ValueError, TypeError, AttributeError):
pass
# Assert that litellm.completion was called when using a custom LLM provider
mock_completion.assert_called_once()
@ -296,7 +296,6 @@ async def test_bedrock_converse_budget_tokens_preserved():
mock_acompletion.assert_called_once()
call_kwargs = mock_acompletion.call_args.kwargs
print("acompletion call kwargs: ", json.dumps(call_kwargs, indent=4, default=str))
# Verify thinking parameter is passed through with budget_tokens preserved
thinking_param = call_kwargs.get("thinking")
@ -328,8 +327,8 @@ def test_openai_model_with_thinking_converts_to_reasoning():
api_key="test-api-key",
thinking={"type": "enabled", "budget_tokens": 1024},
)
except (ValueError, TypeError, AttributeError) as e:
print(f"Error: {e}")
except (ValueError, TypeError, AttributeError):
pass
mock_responses.assert_called_once()
@ -1438,3 +1437,93 @@ async def test_anthropic_messages_leaves_non_provider_failures_unmapped():
)
assert "Traceback" not in str(excinfo.value)
@pytest.mark.asyncio
async def test_anthropic_pass_through_drop_params(monkeypatch):
from litellm.llms.anthropic.experimental_pass_through.messages import handler
mock_handler = AsyncMock()
monkeypatch.setattr(handler.base_llm_http_handler, "anthropic_messages_handler", mock_handler)
await handler.anthropic_messages_handler(
model="vertex_ai/claude-3-haiku-20240307",
messages=[{"role": "user", "content": "hello"}],
max_tokens=64,
drop_params=True,
context_management={"edits": []},
custom_llm_provider="vertex_ai",
is_async=True,
)
optional_params = mock_handler.call_args.kwargs.get("anthropic_messages_optional_request_params", {})
assert "context_management" not in optional_params
def test_drop_params_filters_unsupported_anthropic_params():
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
_drop_unsupported_anthropic_messages_params,
)
params = {
"max_tokens": 64,
"metadata": {"request_id": "test"},
"thinking": {"type": "enabled", "budget_tokens": 1024},
"output_config": {"effort": "high"},
"context_management": {"edits": []},
}
filtered = _drop_unsupported_anthropic_messages_params(
anthropic_messages_optional_request_params=params,
model="claude-3-haiku-20240307",
custom_llm_provider="vertex_ai",
additional_drop_params=["metadata"],
)
assert filtered == {"max_tokens": 64}
assert params["thinking"] == {"type": "enabled", "budget_tokens": 1024}
assert "context_management" in params
def test_drop_params_preserves_supported_anthropic_params():
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
_drop_unsupported_anthropic_messages_params,
)
params = {
"thinking": {"type": "enabled", "budget_tokens": 1024},
"output_config": {"effort": "high"},
"context_management": {"edits": []},
}
filtered = _drop_unsupported_anthropic_messages_params(
anthropic_messages_optional_request_params=params,
model="claude-opus-4-5-20251101",
custom_llm_provider="vertex_ai",
)
assert filtered == params
assert filtered is not params
@pytest.mark.asyncio
async def test_anthropic_pass_through_keeps_supported_params_without_drop(monkeypatch):
import litellm
from litellm.llms.anthropic.experimental_pass_through.messages import handler
mock_handler = AsyncMock()
monkeypatch.setattr(litellm, "drop_params", False)
monkeypatch.setattr(handler.base_llm_http_handler, "anthropic_messages_handler", mock_handler)
await handler.anthropic_messages_handler(
model="vertex_ai/claude-opus-4-5-20251101",
messages=[{"role": "user", "content": "hello"}],
max_tokens=64,
thinking={"type": "enabled", "budget_tokens": 1024},
drop_params=False,
custom_llm_provider="vertex_ai",
is_async=True,
)
optional_params = mock_handler.call_args.kwargs["anthropic_messages_optional_request_params"]
assert optional_params["thinking"] == {"type": "enabled", "budget_tokens": 1024}