mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge a7837717ac into ee1a6407cb
This commit is contained in:
commit
69a740d228
2 changed files with 161 additions and 17 deletions
|
|
@ -7,8 +7,9 @@
|
|||
|
||||
import asyncio
|
||||
import contextvars
|
||||
from collections.abc import AsyncIterator, Coroutine, Iterator
|
||||
from collections.abc import AsyncIterator, Coroutine, Iterator, Mapping, Sequence
|
||||
from functools import partial
|
||||
from types import MappingProxyType
|
||||
from typing import Any, Final, cast
|
||||
|
||||
import litellm
|
||||
|
|
@ -418,6 +419,32 @@ def validate_anthropic_api_metadata(metadata: dict | None = None) -> dict | None
|
|||
return anthropic_metadata_obj.model_dump(exclude_none=True)
|
||||
|
||||
|
||||
def _drop_unsupported_anthropic_messages_params(
|
||||
anthropic_messages_optional_request_params: Mapping[str, Any],
|
||||
model: str,
|
||||
custom_llm_provider: str | None,
|
||||
additional_drop_params: Sequence[str] | None = None,
|
||||
) -> Mapping[str, Any]:
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
|
||||
additional_drop_params_list: Final = additional_drop_params if isinstance(additional_drop_params, list) else None
|
||||
supports_effort: Final = AnthropicConfig._model_supports_effort_param(model, custom_llm_provider or "anthropic") # pyright: ignore[reportPrivateUsage] # shared Anthropic capability gate
|
||||
supports_reasoning: Final = AnthropicConfig._supports_model_capability( # pyright: ignore[reportPrivateUsage] # shared Anthropic capability gate
|
||||
model, "supports_reasoning", custom_llm_provider or "anthropic"
|
||||
)
|
||||
drop_context_management: Final = custom_llm_provider in ("vertex_ai", "bedrock") and "haiku" in model.lower()
|
||||
return MappingProxyType(
|
||||
{
|
||||
k: v
|
||||
for k, v in anthropic_messages_optional_request_params.items()
|
||||
if not (additional_drop_params_list is not None and k in additional_drop_params_list)
|
||||
and (k != "output_config" or supports_effort)
|
||||
and (k != "thinking" or supports_reasoning)
|
||||
and (k != "context_management" or not drop_context_management)
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def anthropic_messages_handler(
|
||||
max_tokens: int,
|
||||
messages: list[dict],
|
||||
|
|
@ -641,19 +668,47 @@ def anthropic_messages_handler(
|
|||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
)
|
||||
if is_reasoning_auto_summary_enabled():
|
||||
thinking_param: Final = anthropic_messages_optional_request_params.get("thinking")
|
||||
if isinstance(thinking_param, dict) and thinking_param.get("type") != "disabled":
|
||||
anthropic_messages_optional_request_params["thinking"] = {
|
||||
**thinking_param,
|
||||
"display": "summarized",
|
||||
|
||||
should_drop_params: Final = (
|
||||
litellm.drop_params is True
|
||||
or getattr(litellm_params, "drop_params", None) is True
|
||||
or kwargs.get("drop_params") is True
|
||||
)
|
||||
|
||||
filtered_anthropic_messages_params: Final = (
|
||||
_drop_unsupported_anthropic_messages_params(
|
||||
anthropic_messages_optional_request_params=anthropic_messages_optional_request_params,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
additional_drop_params=kwargs.get("additional_drop_params"),
|
||||
)
|
||||
if should_drop_params
|
||||
else anthropic_messages_optional_request_params
|
||||
)
|
||||
thinking_param: Final = filtered_anthropic_messages_params.get("thinking")
|
||||
final_anthropic_messages_params: Final = (
|
||||
MappingProxyType(
|
||||
{
|
||||
**filtered_anthropic_messages_params,
|
||||
"thinking": { # mutable-ok: construct the summarized payload before freezing
|
||||
**thinking_param,
|
||||
"display": "summarized",
|
||||
},
|
||||
}
|
||||
)
|
||||
if is_reasoning_auto_summary_enabled()
|
||||
and isinstance(thinking_param, dict)
|
||||
and thinking_param.get("type") != "disabled"
|
||||
else filtered_anthropic_messages_params
|
||||
)
|
||||
|
||||
return base_llm_http_handler.anthropic_messages_handler(
|
||||
model=model,
|
||||
messages=strip_provider_specific_fields_from_anthropic_messages(messages),
|
||||
anthropic_messages_provider_config=anthropic_messages_provider_config,
|
||||
anthropic_messages_optional_request_params=dict(anthropic_messages_optional_request_params),
|
||||
anthropic_messages_optional_request_params=dict( # mutable-ok: downstream handler requires a dict
|
||||
final_anthropic_messages_params
|
||||
),
|
||||
_is_async=is_async,
|
||||
client=client,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
|
|
|
|||
|
|
@ -42,8 +42,8 @@ def test_anthropic_experimental_pass_through_messages_handler():
|
|||
model="openai/claude-3-5-sonnet-20240620",
|
||||
api_key="test-api-key",
|
||||
)
|
||||
except (ValueError, TypeError, AttributeError) as e:
|
||||
print(f"Error: {e}")
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
pass
|
||||
mock_responses.assert_called_once()
|
||||
assert mock_responses.call_args.kwargs["api_key"] == "test-api-key"
|
||||
|
||||
|
|
@ -129,8 +129,8 @@ def test_anthropic_experimental_pass_through_messages_handler_dynamic_api_key_an
|
|||
api_base="test-api-base",
|
||||
custom_key="custom_value",
|
||||
)
|
||||
except (ValueError, TypeError, AttributeError) as e:
|
||||
print(f"Error: {e}")
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
pass
|
||||
mock_completion.assert_called_once()
|
||||
assert mock_completion.call_args.kwargs["api_key"] == "test-api-key"
|
||||
assert mock_completion.call_args.kwargs["api_base"] == "test-api-base"
|
||||
|
|
@ -244,8 +244,8 @@ def test_anthropic_experimental_pass_through_messages_handler_custom_llm_provide
|
|||
custom_llm_provider="my-custom-llm",
|
||||
api_key="test-api-key",
|
||||
)
|
||||
except (ValueError, TypeError, AttributeError) as e:
|
||||
print(f"Error: {e}")
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
pass
|
||||
|
||||
# Assert that litellm.completion was called when using a custom LLM provider
|
||||
mock_completion.assert_called_once()
|
||||
|
|
@ -296,7 +296,6 @@ async def test_bedrock_converse_budget_tokens_preserved():
|
|||
mock_acompletion.assert_called_once()
|
||||
|
||||
call_kwargs = mock_acompletion.call_args.kwargs
|
||||
print("acompletion call kwargs: ", json.dumps(call_kwargs, indent=4, default=str))
|
||||
|
||||
# Verify thinking parameter is passed through with budget_tokens preserved
|
||||
thinking_param = call_kwargs.get("thinking")
|
||||
|
|
@ -328,8 +327,8 @@ def test_openai_model_with_thinking_converts_to_reasoning():
|
|||
api_key="test-api-key",
|
||||
thinking={"type": "enabled", "budget_tokens": 1024},
|
||||
)
|
||||
except (ValueError, TypeError, AttributeError) as e:
|
||||
print(f"Error: {e}")
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
pass
|
||||
|
||||
mock_responses.assert_called_once()
|
||||
|
||||
|
|
@ -1438,3 +1437,93 @@ async def test_anthropic_messages_leaves_non_provider_failures_unmapped():
|
|||
)
|
||||
|
||||
assert "Traceback" not in str(excinfo.value)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_pass_through_drop_params(monkeypatch):
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
|
||||
mock_handler = AsyncMock()
|
||||
monkeypatch.setattr(handler.base_llm_http_handler, "anthropic_messages_handler", mock_handler)
|
||||
|
||||
await handler.anthropic_messages_handler(
|
||||
model="vertex_ai/claude-3-haiku-20240307",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
max_tokens=64,
|
||||
drop_params=True,
|
||||
context_management={"edits": []},
|
||||
custom_llm_provider="vertex_ai",
|
||||
is_async=True,
|
||||
)
|
||||
|
||||
optional_params = mock_handler.call_args.kwargs.get("anthropic_messages_optional_request_params", {})
|
||||
assert "context_management" not in optional_params
|
||||
|
||||
|
||||
def test_drop_params_filters_unsupported_anthropic_params():
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
|
||||
_drop_unsupported_anthropic_messages_params,
|
||||
)
|
||||
|
||||
params = {
|
||||
"max_tokens": 64,
|
||||
"metadata": {"request_id": "test"},
|
||||
"thinking": {"type": "enabled", "budget_tokens": 1024},
|
||||
"output_config": {"effort": "high"},
|
||||
"context_management": {"edits": []},
|
||||
}
|
||||
|
||||
filtered = _drop_unsupported_anthropic_messages_params(
|
||||
anthropic_messages_optional_request_params=params,
|
||||
model="claude-3-haiku-20240307",
|
||||
custom_llm_provider="vertex_ai",
|
||||
additional_drop_params=["metadata"],
|
||||
)
|
||||
|
||||
assert filtered == {"max_tokens": 64}
|
||||
assert params["thinking"] == {"type": "enabled", "budget_tokens": 1024}
|
||||
assert "context_management" in params
|
||||
|
||||
|
||||
def test_drop_params_preserves_supported_anthropic_params():
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
|
||||
_drop_unsupported_anthropic_messages_params,
|
||||
)
|
||||
|
||||
params = {
|
||||
"thinking": {"type": "enabled", "budget_tokens": 1024},
|
||||
"output_config": {"effort": "high"},
|
||||
"context_management": {"edits": []},
|
||||
}
|
||||
|
||||
filtered = _drop_unsupported_anthropic_messages_params(
|
||||
anthropic_messages_optional_request_params=params,
|
||||
model="claude-opus-4-5-20251101",
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
|
||||
assert filtered == params
|
||||
assert filtered is not params
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_pass_through_keeps_supported_params_without_drop(monkeypatch):
|
||||
import litellm
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
|
||||
mock_handler = AsyncMock()
|
||||
monkeypatch.setattr(litellm, "drop_params", False)
|
||||
monkeypatch.setattr(handler.base_llm_http_handler, "anthropic_messages_handler", mock_handler)
|
||||
|
||||
await handler.anthropic_messages_handler(
|
||||
model="vertex_ai/claude-opus-4-5-20251101",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
max_tokens=64,
|
||||
thinking={"type": "enabled", "budget_tokens": 1024},
|
||||
drop_params=False,
|
||||
custom_llm_provider="vertex_ai",
|
||||
is_async=True,
|
||||
)
|
||||
|
||||
optional_params = mock_handler.call_args.kwargs["anthropic_messages_optional_request_params"]
|
||||
assert optional_params["thinking"] == {"type": "enabled", "budget_tokens": 1024}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue