fix(merge): resolve conflict with main in cost_tracking_settings

PR #22890 used cast(str, ...) / cast(Optional[str], ...) for the return
statements; this PR's approach uses str() for explicit runtime coercion
(addressing Greptile's concern). Keep the str() version and drop the
now-unused cast import.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Julio Quinteros 2026-03-05 14:01:07 -03:00
commit 023794ba62
42 changed files with 1259 additions and 163 deletions

View file

@ -0,0 +1,120 @@
# v1/messages → /responses Parameter Mapping
When you send a request to `/v1/messages` targeting an OpenAI or Azure model, LiteLLM internally routes it through the OpenAI Responses API. This page documents exactly how every parameter gets translated in both directions.
The transformation lives in `litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py`.
## Request: Anthropic → Responses API
### Top-level parameters
| Anthropic (`/v1/messages`) | Responses API | Notes |
|---|---|---|
| `model` | `model` | Passed through as-is |
| `messages` | `input` | Structurally transformed — see the messages section below |
| `system` (string) | `instructions` | Passed as a plain string |
| `system` (list of content blocks) | `instructions` | Text blocks are joined with `\n`; non-text blocks are ignored |
| `max_tokens` | `max_output_tokens` | Renamed |
| `temperature` | `temperature` | Passed through as-is |
| `top_p` | `top_p` | Passed through as-is |
| `tools` | `tools` | Format-translated — see the tools section below |
| `tool_choice` | `tool_choice` | Type-remapped — see the tool_choice section below |
| `thinking` | `reasoning` | Budget tokens mapped to effort level — see the thinking section below |
| `output_format` or `output_config.format` | `text` | Wrapped as `{"format": {"type": "json_schema", "name": "structured_output", "schema": ..., "strict": true}}` |
| `context_management` | `context_management` | Converted from Anthropic dict to OpenAI array format — see the context_management section below |
| `metadata.user_id` | `user` | Extracted from the metadata object and truncated to 64 characters |
| `stop_sequences` | ❌ Not mapped | Dropped silently |
| `top_k` | ❌ Not mapped | Dropped silently |
| `speed` | ❌ Not mapped | Only used to set Anthropic beta headers on the native path |
### How messages get converted
Each Anthropic message is expanded into one or more Responses API input items. The key difference is that `tool_result` and `tool_use` blocks become **top-level items** in the input array rather than being nested inside a message.
| Anthropic message | Responses API input item |
|---|---|
| `user` role, string content | `{"type": "message", "role": "user", "content": [{"type": "input_text", "text": "..."}]}` |
| `user` role, `{"type": "text"}` block | `{"type": "input_text", "text": "..."}` inside a user message |
| `user` role, `{"type": "image", "source": {"type": "base64"}}` | `{"type": "input_image", "image_url": "data:<media_type>;base64,<data>"}` inside a user message |
| `user` role, `{"type": "image", "source": {"type": "url"}}` | `{"type": "input_image", "image_url": "<url>"}` inside a user message |
| `user` role, `{"type": "tool_result"}` block | Top-level `{"type": "function_call_output", "call_id": "...", "output": "..."}` — pulled out of the message entirely |
| `assistant` role, string content | `{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "..."}]}` |
| `assistant` role, `{"type": "text"}` block | `{"type": "output_text", "text": "..."}` inside an assistant message |
| `assistant` role, `{"type": "tool_use"}` block | Top-level `{"type": "function_call", "call_id": "<id>", "name": "...", "arguments": "<JSON string>"}` — pulled out of the message entirely |
| `assistant` role, `{"type": "thinking"}` block | `{"type": "output_text", "text": "<thinking text>"}` inside an assistant message |
### tools
| Anthropic tool | Responses API tool |
|---|---|
| Any tool where `type` starts with `"web_search"` or `name == "web_search"` | `{"type": "web_search_preview"}` |
| All other tools | `{"type": "function", "name": "...", "description": "...", "parameters": <input_schema>}` |
### tool_choice
| Anthropic `tool_choice.type` | Responses API `tool_choice` |
|---|---|
| `"auto"` | `{"type": "auto"}` |
| `"any"` | `{"type": "required"}` |
| `"tool"` | `{"type": "function", "name": "<tool name>"}` |
### thinking → reasoning
The `budget_tokens` value is mapped to a string effort level. `summary` is always set to `"detailed"`.
| `thinking.budget_tokens` | `reasoning.effort` |
|---|---|
| >= 10000 | `"high"` |
| >= 5000 | `"medium"` |
| >= 2000 | `"low"` |
| < 2000 | `"minimal"` |
If `thinking.type` is anything other than `"enabled"`, the `reasoning` field is not sent at all.
### context_management
Anthropic uses a nested dict with an `edits` array. OpenAI uses a flat array of compaction objects.
```
Anthropic input:
{
"edits": [
{
"type": "compact_20260112",
"trigger": {"type": "input_tokens", "value": 150000}
}
]
}
Responses API output:
[
{"type": "compaction", "compact_threshold": 150000}
]
```
## Response: Responses API → Anthropic
When the Responses API reply comes back, LiteLLM converts it into an Anthropic `AnthropicMessagesResponse`.
| Responses API field | Anthropic response field | Notes |
|---|---|---|
| `response.id` | `id` | |
| `response.model` | `model` | Falls back to `"unknown-model"` if missing |
| `ResponseReasoningItem` — `summary[*].text` | `content` block `{"type": "thinking", "thinking": "..."}` | Each non-empty summary text becomes a thinking block |
| `ResponseOutputMessage` — `content[*]` where `type == "output_text"` | `content` block `{"type": "text", "text": "..."}` | |
| `ResponseFunctionToolCall` — `{call_id, name, arguments}` | `content` block `{"type": "tool_use", "id": "...", "name": "...", "input": {...}}` | `arguments` is JSON-parsed back into a dict |
| Any `function_call` present in output | `stop_reason: "tool_use"` | |
| `response.status == "incomplete"` | `stop_reason: "max_tokens"` | Takes precedence over the default |
| Everything else | `stop_reason: "end_turn"` | Default |
| `response.usage.input_tokens` | `usage.input_tokens` | |
| `response.usage.output_tokens` | `usage.output_tokens` | |
| *(hardcoded)* | `type: "message"` | Always set |
| *(hardcoded)* | `role: "assistant"` | Always set |
| *(hardcoded)* | `stop_sequence: null` | Always null on this path |

View file

@ -0,0 +1,157 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Amazon Bedrock Mantle
[Amazon Bedrock Mantle](https://docs.aws.amazon.com/bedrock/latest/userguide/bedrock-mantle.html) is Amazon Bedrock's distributed inference engine (Project Mantle) that exposes an **OpenAI-compatible API** for Bedrock-hosted models.
Use this provider to call Bedrock Mantle models with accurate **AWS Bedrock pricing** instead of OpenAI pricing.
:::tip
**We support ALL Bedrock Mantle models, just set `model=bedrock_mantle/<model-id>` as a prefix when sending litellm requests**
:::
## API Key
```python
# env variable
os.environ['BEDROCK_MANTLE_API_KEY'] = "your-aws-bedrock-api-key"
# optional: override region (defaults to us-east-1)
os.environ['BEDROCK_MANTLE_REGION'] = "us-east-1" # or use AWS_REGION
```
## Supported Models
| Model | Context Window | Input (per 1M tokens) | Output (per 1M tokens) |
|-------|---------------|----------------------|------------------------|
| `openai.gpt-oss-120b` | 131K | $0.15 | $0.60 |
| `openai.gpt-oss-20b` | 131K | $0.075 | $0.30 |
| `openai.gpt-oss-safeguard-120b` | 131K | $0.15 | $0.60 |
| `openai.gpt-oss-safeguard-20b` | 131K | $0.075 | $0.30 |
## Sample Usage
<Tabs>
<TabItem value="sdk" label="SDK">
```python
from litellm import completion
import os
os.environ['BEDROCK_MANTLE_API_KEY'] = "your-bedrock-api-key"
response = completion(
model="bedrock_mantle/openai.gpt-oss-120b",
messages=[{"role": "user", "content": "hello from litellm"}],
)
print(response)
```
</TabItem>
<TabItem value="streaming" label="Streaming">
```python
from litellm import completion
import os
os.environ['BEDROCK_MANTLE_API_KEY'] = "your-bedrock-api-key"
response = completion(
model="bedrock_mantle/openai.gpt-oss-120b",
messages=[{"role": "user", "content": "hello from litellm"}],
stream=True,
)
for chunk in response:
print(chunk)
```
</TabItem>
<TabItem value="async" label="Async">
```python
import asyncio
from litellm import acompletion
import os
os.environ['BEDROCK_MANTLE_API_KEY'] = "your-bedrock-api-key"
async def main():
response = await acompletion(
model="bedrock_mantle/openai.gpt-oss-120b",
messages=[{"role": "user", "content": "hello from litellm"}],
)
print(response)
asyncio.run(main())
```
</TabItem>
</Tabs>
## Region Configuration
The API base URL is `https://bedrock-mantle.{region}.api.aws/v1`. Region is resolved in this order:
1. `BEDROCK_MANTLE_REGION` env var
2. `AWS_REGION` env var
3. Default: `us-east-1`
**Supported regions:** `us-east-1`, `us-east-2`, `us-west-2`, `eu-west-1`, `eu-west-2`, `eu-central-1`, `eu-south-1`, `eu-north-1`, `ap-northeast-1`, `ap-south-1`, `ap-southeast-3`, `sa-east-1`
```python
import os
os.environ['BEDROCK_MANTLE_REGION'] = "eu-west-1"
# or pass api_base directly
response = completion(
model="bedrock_mantle/openai.gpt-oss-120b",
messages=[{"role": "user", "content": "hello"}],
api_base="https://bedrock-mantle.eu-west-1.api.aws/v1",
)
```
## Usage with LiteLLM Proxy
### 1. Set Bedrock Mantle models on config.yaml
```yaml
model_list:
- model_name: gpt-oss-120b
litellm_params:
model: bedrock_mantle/openai.gpt-oss-120b
api_key: os.environ/BEDROCK_MANTLE_API_KEY
# optional region override:
api_base: "https://bedrock-mantle.us-east-1.api.aws/v1"
- model_name: gpt-oss-20b
litellm_params:
model: bedrock_mantle/openai.gpt-oss-20b
api_key: os.environ/BEDROCK_MANTLE_API_KEY
```
### 2. Start the proxy
```shell
litellm --config /path/to/config.yaml
```
### 3. Send a request
```python
import openai
client = openai.OpenAI(
api_key="anything",
base_url="http://0.0.0.0:4000",
)
response = client.chat.completions.create(
model="gpt-oss-120b",
messages=[{"role": "user", "content": "hello from litellm"}],
)
print(response)
```

View file

@ -624,6 +624,7 @@ const sidebars = {
items: [
"anthropic_unified/index",
"anthropic_unified/structured_output",
"anthropic_unified/messages_to_responses_mapping",
]
},
"anthropic_count_tokens",
@ -795,6 +796,7 @@ const sidebars = {
"providers/bedrock_realtime_with_audio",
"providers/aws_polly",
"providers/bedrock_vector_store",
"providers/bedrock_mantle",
]
},
"providers/litellm_proxy",

View file

@ -593,6 +593,7 @@ minimax_models: Set = set()
aws_polly_models: Set = set()
gigachat_models: Set = set()
llamagate_models: Set = set()
bedrock_mantle_models: Set = set()
def is_bedrock_pricing_only_model(key: str) -> bool:
@ -855,6 +856,8 @@ def add_known_models(model_cost_map: Optional[Dict] = None):
gigachat_models.add(key)
elif value.get("litellm_provider") == "llamagate":
llamagate_models.add(key)
elif value.get("litellm_provider") == "bedrock_mantle":
bedrock_mantle_models.add(key)
add_known_models()
@ -962,6 +965,7 @@ model_list = list(
| ovhcloud_models
| lemonade_models
| docker_model_runner_models
| bedrock_mantle_models
| set(clarifai_models)
)
@ -1065,6 +1069,7 @@ models_by_provider: dict = {
"aws_polly": aws_polly_models,
"gigachat": gigachat_models,
"llamagate": llamagate_models,
"bedrock_mantle": bedrock_mantle_models
}
# mapping for those models which have larger equivalents
@ -1426,6 +1431,7 @@ if TYPE_CHECKING:
from .llms.topaz.image_variations.transformation import TopazImageVariationConfig as TopazImageVariationConfig
from litellm.llms.openai.completion.transformation import OpenAITextCompletionConfig as OpenAITextCompletionConfig
from .llms.groq.chat.transformation import GroqChatConfig as GroqChatConfig
from .llms.bedrock_mantle.chat.transformation import BedrockMantleChatConfig as BedrockMantleChatConfig
from .llms.a2a.chat.transformation import A2AConfig as A2AConfig
from .llms.voyage.embedding.transformation import VoyageEmbeddingConfig as VoyageEmbeddingConfig
from .llms.voyage.embedding.transformation_contextual import VoyageContextualEmbeddingConfig as VoyageContextualEmbeddingConfig

View file

@ -214,6 +214,7 @@ LLM_CONFIG_NAMES = (
"TopazImageVariationConfig",
"OpenAITextCompletionConfig",
"GroqChatConfig",
"BedrockMantleChatConfig",
"A2AConfig",
"GenAIHubOrchestrationConfig",
"VoyageEmbeddingConfig",
@ -858,6 +859,7 @@ _LLM_CONFIGS_IMPORT_MAP = {
"OpenAITextCompletionConfig",
),
"GroqChatConfig": (".llms.groq.chat.transformation", "GroqChatConfig"),
"BedrockMantleChatConfig": (".llms.bedrock_mantle.chat.transformation", "BedrockMantleChatConfig"),
"A2AConfig": (".llms.a2a.chat.transformation", "A2AConfig"),
"GenAIHubOrchestrationConfig": (
".llms.sap.chat.transformation",

View file

@ -162,6 +162,49 @@ async def _send_message_via_completion_bridge(
return LiteLLMSendMessageResponse.from_dict(response_dict)
async def _execute_a2a_send_with_retry(
a2a_client: Any,
request: Any,
agent_card: Any,
card_url: Optional[str],
api_base: Optional[str],
agent_name: Optional[str],
) -> Any:
"""Send an A2A message with retry logic for localhost URL errors."""
a2a_response = None
for _ in range(2): # max 2 attempts: original + 1 retry
try:
a2a_response = await a2a_client.send_message(request)
break # success, exit retry loop
except A2ALocalhostURLError as e:
a2a_client = handle_a2a_localhost_retry(
error=e,
agent_card=agent_card,
a2a_client=a2a_client,
is_streaming=False,
)
card_url = agent_card.url if agent_card else None
except Exception as e:
try:
map_a2a_exception(e, card_url, api_base, model=agent_name)
except A2ALocalhostURLError as localhost_err:
a2a_client = handle_a2a_localhost_retry(
error=localhost_err,
agent_card=agent_card,
a2a_client=a2a_client,
is_streaming=False,
)
card_url = agent_card.url if agent_card else None
continue
except Exception:
raise
if a2a_response is None:
raise RuntimeError(
"A2A send_message failed: no response received after retry attempts."
)
return a2a_response
@client
async def asend_message(
a2a_client: Optional["A2AClientType"] = None,
@ -279,44 +322,17 @@ async def asend_message(
if getattr(message, "context_id", None) is None:
message.context_id = context_id
# Retry loop: if connection fails due to localhost URL in agent card, retry with fixed URL
a2a_response = None
for _ in range(2): # max 2 attempts: original + 1 retry
try:
a2a_response = await a2a_client.send_message(request)
break # success, exit retry loop
except A2ALocalhostURLError as e:
# Localhost URL error - fix and retry
a2a_client = handle_a2a_localhost_retry(
error=e,
agent_card=agent_card,
a2a_client=a2a_client,
is_streaming=False,
)
card_url = agent_card.url if agent_card else None
except Exception as e:
# Map exception - will raise A2ALocalhostURLError if applicable
try:
map_a2a_exception(e, card_url, api_base, model=agent_name)
except A2ALocalhostURLError as localhost_err:
# Localhost URL error - fix and retry
a2a_client = handle_a2a_localhost_retry(
error=localhost_err,
agent_card=agent_card,
a2a_client=a2a_client,
is_streaming=False,
)
card_url = agent_card.url if agent_card else None
continue
except Exception:
# Re-raise the mapped exception
raise
a2a_response = await _execute_a2a_send_with_retry(
a2a_client=a2a_client,
request=request,
agent_card=agent_card,
card_url=card_url,
api_base=api_base,
agent_name=agent_name,
)
verbose_logger.info(f"A2A send_message completed, request_id={request.id}")
# a2a_response is guaranteed to be set if we reach here (loop breaks on success or raises)
assert a2a_response is not None
# Wrap in LiteLLM response type for _hidden_params support
response = LiteLLMSendMessageResponse.from_a2a_response(a2a_response)

View file

@ -33,6 +33,7 @@ from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import (
CancelBatchRequest,
CreateBatchRequest,
FileExpiresAfter,
RetrieveBatchRequest,
)
from litellm.types.router import GenericLiteLLMParams
@ -219,7 +220,7 @@ def create_batch( # noqa: PLR0915
extra_body=extra_body,
)
if output_expires_after is not None:
_create_batch_request["output_expires_after"] = output_expires_after
_create_batch_request["output_expires_after"] = cast(FileExpiresAfter, output_expires_after)
if model is not None:
provider_config = ProviderConfigManager.get_provider_batches_config(
model=model,

View file

@ -126,6 +126,19 @@ async def acreate_fine_tuning_job(
raise e
def _resolve_fine_tuning_timeout(
timeout: Any,
custom_llm_provider: str,
) -> Union[float, httpx.Timeout]:
"""Normalise a raw timeout value to a float (seconds) or httpx.Timeout for fine-tuning calls."""
timeout = timeout or 600.0
if isinstance(timeout, httpx.Timeout):
if not supports_httpx_timeout(custom_llm_provider):
return float(timeout.read or 600)
return timeout
return float(timeout)
@client
def create_fine_tuning_job(
model: str,
@ -164,21 +177,10 @@ def create_fine_tuning_job(
_oai_hyperparameters: Hyperparameters = Hyperparameters(
**hyperparameters
) # Typed Hyperparameters for OpenAI Spec
### TIMEOUT LOGIC ###
timeout = optional_params.timeout or kwargs.get("request_timeout", 600) or 600
# set timeout for 10 minutes by default
if (
timeout is not None
and isinstance(timeout, httpx.Timeout)
and supports_httpx_timeout(custom_llm_provider) is False
):
read_timeout = timeout.read or 600
timeout = read_timeout # default 10 min timeout
elif timeout is not None and not isinstance(timeout, httpx.Timeout):
timeout = float(timeout) # type: ignore
elif timeout is None:
timeout = 600.0
timeout = _resolve_fine_tuning_timeout(
optional_params.timeout or kwargs.get("request_timeout", 600),
custom_llm_provider,
)
# OpenAI
if custom_llm_provider == "openai":

View file

@ -561,6 +561,13 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915
) = litellm.GroqChatConfig()._get_openai_compatible_provider_info(
api_base, api_key
)
elif custom_llm_provider == "bedrock_mantle":
(
api_base,
dynamic_api_key,
) = litellm.BedrockMantleChatConfig()._get_openai_compatible_provider_info(
api_base, api_key
)
elif custom_llm_provider == "nvidia_nim":
# nvidia_nim is openai compatible, we just need to set this to custom_openai and have the api_base be https://api.endpoints.anyscale.com/v1
api_base = (

View file

@ -88,6 +88,8 @@ def get_supported_openai_params( # noqa: PLR0915
return litellm.VolcEngineConfig().get_supported_openai_params(model=model)
elif custom_llm_provider == "groq":
return litellm.GroqChatConfig().get_supported_openai_params(model=model)
elif custom_llm_provider == "bedrock_mantle":
return litellm.BedrockMantleChatConfig().get_supported_openai_params(model=model)
elif custom_llm_provider == "hosted_vllm":
return litellm.HostedVLLMChatConfig().get_supported_openai_params(model=model)
elif custom_llm_provider == "vllm":

View file

@ -194,6 +194,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
"web_search_options",
"speed",
"context_management",
"cache_control",
]
if (
@ -1031,6 +1032,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
elif param == "speed" and isinstance(value, str):
# Pass through Anthropic-specific speed parameter for fast mode
optional_params["speed"] = value
elif param == "cache_control" and isinstance(value, dict):
# Pass through top-level cache_control for automatic prompt caching
optional_params["cache_control"] = value
## handle thinking tokens
self.update_optional_params_with_thinking_tokens(

View file

@ -35,7 +35,7 @@ class AzureBatchesAPI(BaseAzureLLM):
create_batch_data: CreateBatchRequest,
azure_client: Union[AsyncAzureOpenAI, AsyncOpenAI],
) -> LiteLLMBatch:
response = await azure_client.batches.create(**create_batch_data)
response = await azure_client.batches.create(**create_batch_data) # type: ignore[arg-type]
return LiteLLMBatch(**response.model_dump())
def create_batch(
@ -73,7 +73,7 @@ class AzureBatchesAPI(BaseAzureLLM):
return self.acreate_batch( # type: ignore
create_batch_data=create_batch_data, azure_client=azure_client
)
response = cast(Union[AzureOpenAI, OpenAI], azure_client).batches.create(**create_batch_data)
response = cast(Union[AzureOpenAI, OpenAI], azure_client).batches.create(**create_batch_data) # type: ignore[arg-type]
return LiteLLMBatch(**response.model_dump())
async def aretrieve_batch(
@ -81,7 +81,7 @@ class AzureBatchesAPI(BaseAzureLLM):
retrieve_batch_data: RetrieveBatchRequest,
client: Union[AsyncAzureOpenAI, AsyncOpenAI],
) -> LiteLLMBatch:
response = await client.batches.retrieve(**retrieve_batch_data)
response = await client.batches.retrieve(**retrieve_batch_data) # type: ignore[arg-type]
return LiteLLMBatch(**response.model_dump())
def retrieve_batch(

View file

@ -1,7 +1,7 @@
"""
Azure Anthropic messages transformation config - extends AnthropicMessagesConfig with Azure authentication
"""
from typing import TYPE_CHECKING, Any, List, Optional, Tuple
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
@ -114,3 +114,53 @@ class AzureAnthropicMessagesConfig(AnthropicMessagesConfig):
return api_base
def _remove_scope_from_cache_control(
self, anthropic_messages_request: Dict
) -> None:
"""
Remove `scope` field from cache_control for Azure AI Foundry.
Azure AI Foundry's Anthropic endpoint does not support the `scope` field
(e.g., "global" for cross-request caching). Only `type` and `ttl` are supported.
Processes both `system` and `messages` content blocks.
"""
def _sanitize(cache_control: Any) -> None:
if isinstance(cache_control, dict):
cache_control.pop("scope", None)
def _process_content_list(content: list) -> None:
for item in content:
if isinstance(item, dict) and "cache_control" in item:
_sanitize(item["cache_control"])
if "system" in anthropic_messages_request:
system = anthropic_messages_request["system"]
if isinstance(system, list):
_process_content_list(system)
if "messages" in anthropic_messages_request:
for message in anthropic_messages_request["messages"]:
if isinstance(message, dict) and "content" in message:
content = message["content"]
if isinstance(content, list):
_process_content_list(content)
def transform_anthropic_messages_request(
self,
model: str,
messages: List[Dict],
anthropic_messages_optional_request_params: Dict,
litellm_params: GenericLiteLLMParams,
headers: dict,
) -> Dict:
anthropic_messages_request = super().transform_anthropic_messages_request(
model=model,
messages=messages,
anthropic_messages_optional_request_params=anthropic_messages_optional_request_params,
litellm_params=litellm_params,
headers=headers,
)
self._remove_scope_from_cache_control(anthropic_messages_request)
return anthropic_messages_request

View file

@ -118,10 +118,13 @@ class AmazonAnthropicClaudeMessagesConfig(
self, anthropic_messages_request: Dict, model: Optional[str] = None
) -> None:
"""
Remove `ttl` field from cache_control in messages.
Bedrock doesn't support the ttl field in cache_control.
Remove unsupported fields from cache_control for Bedrock.
Update: Bedock supports `5m` and `1h` for Claude 4.5 models.
Bedrock only supports `type` and `ttl` in cache_control. It does NOT support:
- `scope` (e.g., "global") - always removed
- `ttl` - removed for older models; Claude 4.5+ supports "5m" and "1h"
Processes both `system` and `messages` content blocks.
Args:
anthropic_messages_request: The request dictionary to modify in-place
@ -131,23 +134,36 @@ class AmazonAnthropicClaudeMessagesConfig(
if model:
is_claude_4_5 = self._is_claude_4_5_on_bedrock(model)
def _sanitize_cache_control(cache_control: dict) -> None:
if not isinstance(cache_control, dict):
return
# Bedrock doesn't support scope (e.g., "global" for cross-request caching)
cache_control.pop("scope", None)
# Remove ttl for models that don't support it
if "ttl" in cache_control:
ttl = cache_control["ttl"]
if is_claude_4_5 and ttl in ["5m", "1h"]:
return
cache_control.pop("ttl", None)
def _process_content_list(content: list) -> None:
for item in content:
if isinstance(item, dict) and "cache_control" in item:
_sanitize_cache_control(item["cache_control"])
# Process system (list of content blocks)
if "system" in anthropic_messages_request:
system = anthropic_messages_request["system"]
if isinstance(system, list):
_process_content_list(system)
# Process messages
if "messages" in anthropic_messages_request:
for message in anthropic_messages_request["messages"]:
if isinstance(message, dict) and "content" in message:
content = message["content"]
if isinstance(content, list):
for item in content:
if isinstance(item, dict) and "cache_control" in item:
cache_control = item["cache_control"]
if (
isinstance(cache_control, dict)
and "ttl" in cache_control
):
ttl = cache_control["ttl"]
if is_claude_4_5 and ttl in ["5m", "1h"]:
continue
cache_control.pop("ttl", None)
_process_content_list(content)
def _supports_extended_thinking_on_bedrock(self, model: str) -> bool:
"""

View file

@ -0,0 +1,80 @@
"""
Amazon Bedrock Mantle - OpenAI-compatible inference engine in Amazon Bedrock.
API docs: https://docs.aws.amazon.com/bedrock/latest/userguide/bedrock-mantle.html
Base URL: https://bedrock-mantle.{region}.api.aws/v1
Auth: AWS Bedrock API key as Bearer token (set via BEDROCK_MANTLE_API_KEY env var)
or region-aware key via BEDROCK_MANTLE_{REGION}_API_KEY.
"""
from typing import Iterator, AsyncIterator, Any, Optional, Tuple, Union
import litellm
from litellm._logging import verbose_logger
from litellm.secret_managers.main import get_secret_str
from ...openai_like.chat.transformation import OpenAILikeChatConfig
BEDROCK_MANTLE_DEFAULT_REGION = "us-east-1"
class BedrockMantleChatConfig(OpenAILikeChatConfig):
"""
Transformation config for Amazon Bedrock Mantle OpenAI-compatible API.
"""
@property
def custom_llm_provider(self) -> Optional[str]:
return "bedrock_mantle"
@classmethod
def get_config(cls):
return super().get_config()
def _get_openai_compatible_provider_info(
self, api_base: Optional[str], api_key: Optional[str]
) -> Tuple[Optional[str], Optional[str]]:
region = (
get_secret_str("BEDROCK_MANTLE_REGION")
or get_secret_str("AWS_REGION")
or BEDROCK_MANTLE_DEFAULT_REGION
)
api_base = (
api_base
or get_secret_str("BEDROCK_MANTLE_API_BASE")
or f"https://bedrock-mantle.{region}.api.aws/v1"
)
dynamic_api_key = api_key or get_secret_str("BEDROCK_MANTLE_API_KEY")
return api_base, dynamic_api_key
def get_supported_openai_params(self, model: str) -> list:
base_params = super().get_supported_openai_params(model)
try:
if litellm.supports_reasoning(
model=model, custom_llm_provider=self.custom_llm_provider
):
if "reasoning_effort" not in base_params:
base_params.append("reasoning_effort")
except Exception as e:
verbose_logger.debug(
f"BedrockMantleChatConfig: error checking reasoning support: {e}"
)
return base_params
def get_model_response_iterator(
self,
streaming_response: Union[Iterator[str], AsyncIterator[str], Any],
sync_stream: bool,
json_mode: Optional[bool] = False,
) -> Any:
from litellm.llms.openai.chat.gpt_transformation import (
OpenAIChatCompletionStreamingHandler,
)
return OpenAIChatCompletionStreamingHandler(
streaming_response=streaming_response,
sync_stream=sync_stream,
json_mode=json_mode,
)

View file

@ -1938,7 +1938,7 @@ class OpenAIBatchesAPI(BaseLLM):
create_batch_data: CreateBatchRequest,
openai_client: AsyncOpenAI,
) -> LiteLLMBatch:
response = await openai_client.batches.create(**create_batch_data)
response = await openai_client.batches.create(**create_batch_data) # type: ignore[arg-type]
return LiteLLMBatch(**response.model_dump())
def create_batch(
@ -1974,7 +1974,7 @@ class OpenAIBatchesAPI(BaseLLM):
return self.acreate_batch( # type: ignore
create_batch_data=create_batch_data, openai_client=openai_client
)
response = cast(OpenAI, openai_client).batches.create(**create_batch_data)
response = cast(OpenAI, openai_client).batches.create(**create_batch_data) # type: ignore[arg-type]
return LiteLLMBatch(**response.model_dump())
@ -1984,7 +1984,7 @@ class OpenAIBatchesAPI(BaseLLM):
openai_client: AsyncOpenAI,
) -> LiteLLMBatch:
verbose_logger.debug("retrieving batch, args= %s", retrieve_batch_data)
response = await openai_client.batches.retrieve(**retrieve_batch_data)
response = await openai_client.batches.retrieve(**retrieve_batch_data) # type: ignore[arg-type]
return LiteLLMBatch(**response.model_dump())
def retrieve_batch(
@ -2020,7 +2020,7 @@ class OpenAIBatchesAPI(BaseLLM):
return self.aretrieve_batch( # type: ignore
retrieve_batch_data=retrieve_batch_data, openai_client=openai_client
)
response = cast(OpenAI, openai_client).batches.retrieve(**retrieve_batch_data)
response = cast(OpenAI, openai_client).batches.retrieve(**retrieve_batch_data) # type: ignore[arg-type]
return LiteLLMBatch(**response.model_dump())
async def acancel_batch(

View file

@ -91,9 +91,9 @@ class OpenRouterImageEditConfig(BaseImageEditConfig):
if key == "size":
if "image_config" not in mapped_params:
mapped_params["image_config"] = {}
mapped_params["image_config"]["aspect_ratio"] = self._map_size_to_aspect_ratio(value)
mapped_params["image_config"]["aspect_ratio"] = self._map_size_to_aspect_ratio(cast(str, value))
elif key == "quality":
image_size = self._map_quality_to_image_size(value)
image_size = self._map_quality_to_image_size(cast(str, value))
if image_size:
if "image_config" not in mapped_params:
mapped_params["image_config"] = {}

View file

@ -3,7 +3,7 @@ Calls SearchAPI.io's Google Search API endpoint.
SearchAPI.io API Reference: https://www.searchapi.io/docs/google
"""
from typing import Dict, List, Literal, Optional, TypedDict, Union
from typing import Dict, List, Literal, Optional, TypedDict, Union, cast
from urllib.parse import urlencode
import httpx
@ -164,7 +164,7 @@ class SearchAPIConfig(BaseSearchConfig):
if "country" in optional_params:
# Map to gl parameter
result_data["gl"] = optional_params["country"].lower()
result_data["gl"] = cast(str, optional_params["country"]).lower()
# Pass through all other SearchAPI.io-specific parameters
for param, value in optional_params.items():

View file

@ -595,6 +595,8 @@ def _transform_request_body(
safety_settings: Optional[List[SafetSettingsConfig]] = optional_params.pop(
"safety_settings", None
) # type: ignore
# Drop output_config as it's not supported by Vertex AI
optional_params.pop("output_config", None)
config_fields = GenerationConfig.__annotations__.keys()
# If the LiteLLM client sends Gemini-supported parameter "labels", add it

View file

@ -152,4 +152,8 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert
"output_format", None
) # do not pass output_format in request body to vertex ai - vertex ai does not support output_format as yet
anthropic_messages_request.pop(
"output_config", None
) # do not pass output_config in request body to vertex ai - vertex ai does not support output_config
return anthropic_messages_request

View file

@ -107,6 +107,9 @@ class VertexAIAnthropicConfig(AnthropicConfig):
# VertexAI doesn't support output_format parameter, remove it if present
data.pop("output_format", None)
# VertexAI doesn't support output_config parameter, remove it if present
data.pop("output_config", None)
tools = optional_params.get("tools")
tool_search_used = self.is_tool_search_used(tools)

View file

@ -2241,6 +2241,32 @@ def completion( # type: ignore # noqa: PLR0915
logging_obj=logging, # model call logging done inside the class as we make need to modify I/O to fit aleph alpha's requirements
client=client,
)
elif custom_llm_provider == "bedrock_mantle":
api_base = api_base or litellm.api_base or get_secret("BEDROCK_MANTLE_API_BASE")
api_key = api_key or litellm.api_key or get_secret("BEDROCK_MANTLE_API_KEY")
headers = headers or litellm.headers
config = litellm.BedrockMantleChatConfig.get_config()
for k, v in config.items():
if k not in optional_params:
optional_params[k] = v
response = base_llm_http_handler.completion(
model=model,
stream=stream,
messages=messages,
acompletion=acompletion,
api_base=api_base,
model_response=model_response,
optional_params=optional_params,
litellm_params=litellm_params,
shared_session=shared_session,
custom_llm_provider=custom_llm_provider,
timeout=timeout,
headers=headers,
encoding=_get_encoding(),
api_key=api_key,
logging_obj=logging,
client=client,
)
elif custom_llm_provider == "a2a":
# A2A (Agent-to-Agent) Protocol
# Resolve agent configuration from registry if model format is "a2a/<agent-name>"

View file

@ -5817,6 +5817,15 @@
],
"source": "https://devblogs.microsoft.com/foundry/whats-new-in-azure-ai-foundry-august-2025/#mistral-document-ai-(ocr)-%E2%80%94-serverless-in-foundry"
},
"azure_ai/mistral-document-ai-2512": {
"litellm_provider": "azure_ai",
"ocr_cost_per_page": 0.003,
"mode": "ocr",
"supported_endpoints": [
"/v1/ocr"
],
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/"
},
"azure_ai/doc-intelligence/prebuilt-read": {
"litellm_provider": "azure_ai",
"ocr_cost_per_page": 0.0015,
@ -39140,5 +39149,59 @@
"metadata": {
"notes": "DuckDuckGo Instant Answer API is free and does not require an API key."
}
},
"bedrock_mantle/openai.gpt-oss-120b": {
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 6e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"bedrock_mantle/openai.gpt-oss-20b": {
"input_cost_per_token": 7.5e-08,
"output_cost_per_token": 3e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"bedrock_mantle/openai.gpt-oss-safeguard-120b": {
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 6e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"bedrock_mantle/openai.gpt-oss-safeguard-20b": {
"input_cost_per_token": 7.5e-08,
"output_cost_per_token": 3e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
}
}

View file

@ -145,8 +145,12 @@ def _safe_get_request_headers(request: Optional[Request]) -> dict:
return {}
state = getattr(request, "state", None)
cached = getattr(state, "_cached_headers", None)
if cached is not None:
if isinstance(cached, dict):
return cached
if cached is not None:
verbose_proxy_logger.debug(
"Unexpected cached request headers type - {}".format(type(cached))
)
try:
headers = dict(request.headers)
except Exception as e:
@ -516,4 +520,3 @@ def _add_vector_store_id_from_path(request_data: dict, request: Request) -> None
verbose_proxy_logger.debug(
f"populate_request_with_path_params: No vector_store_id present in path={path}"
)

View file

@ -133,7 +133,7 @@ class SpendLogCleanup:
if self.pod_lock_manager and self.pod_lock_manager.redis_cache:
lock_acquired = await self.pod_lock_manager.acquire_lock(
cronjob_id=SPEND_LOG_CLEANUP_JOB_NAME,
)
) or False
verbose_proxy_logger.info(
f"Lock acquisition attempt: {'successful' if lock_acquired else 'failed'} at {datetime.now()}"
)

View file

@ -98,7 +98,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
"text:shieldPrompt", cast(dict, request_body)
)
last_response = AzurePromptShieldGuardrailResponse(**response_json)
last_response = cast(AzurePromptShieldGuardrailResponse, response_json)
if last_response["userPromptAnalysis"].get("attackDetected"):
verbose_proxy_logger.warning(

View file

@ -125,13 +125,13 @@ class AzureContentSafetyTextModerationGuardrail(AzureGuardrailBase, CustomGuardr
for chunk in chunks:
request_body = AzureTextModerationGuardrailRequestBody(
text=chunk,
**self.optional_params_request_body,
**self.optional_params_request_body, # type: ignore[misc]
)
response_json = await self._post_to_content_safety(
"text:analyze", cast(dict, request_body)
)
chunk_response = AzureTextModerationGuardrailResponse(**response_json)
chunk_response = cast(AzureTextModerationGuardrailResponse, response_json)
# For multi-chunk texts the callers only see the final response,
# so we must check every intermediate chunk here to avoid silently

View file

@ -312,6 +312,13 @@ class GenericGuardrailAPI(CustomGuardrail):
return_inputs.update(inputs)
return return_inputs
def _build_request_headers(self) -> dict:
"""Build HTTP headers for the guardrail API request."""
headers = {"Content-Type": "application/json"}
if self.headers:
headers.update(self.headers)
return headers
def _build_guardrail_return_inputs(
self,
*,
@ -416,10 +423,7 @@ class GenericGuardrailAPI(CustomGuardrail):
model=model,
)
# Prepare headers
headers = {"Content-Type": "application/json"}
if self.headers:
headers.update(self.headers)
headers = self._build_request_headers()
try:
# Make the API request

View file

@ -799,7 +799,7 @@ class LiteLLMProxyRequestSetup:
Add team-based callbacks from the config
"""
team_config = proxy_config.load_team_config(team_id=team_id)
if len(team_config.keys()) == 0:
if not isinstance(team_config, dict) or len(team_config) == 0:
return None
callback_vars_dict = {**team_config.get("callback_vars", team_config)}

View file

@ -305,7 +305,6 @@ model LiteLLM_MCPServerTable {
registration_url String?
allow_all_keys Boolean @default(false)
available_on_public_internet Boolean @default(true)
spec_path String?
is_byok Boolean @default(false)
byok_description String[] @default([])
byok_api_key_help_url String?

View file

@ -404,74 +404,9 @@ class MCPEnhancedStreamingIterator(BaseResponsesAPIStreamingIterator):
# Phase 1: Initial Response Stream (emit standard OpenAI events first)
if self.phase == "initial_response":
# Create the initial response iterator if not already created
if self.base_iterator is None:
await self._create_initial_response_iterator()
if self.base_iterator is None:
# LLM call failed — still emit MCP discovery events before finishing
if self.mcp_discovery_events:
self.phase = "mcp_discovery"
else:
self.phase = "finished"
raise StopAsyncIteration
if self.base_iterator:
# Check if base_iterator is actually iterable
if hasattr(self.base_iterator, "__anext__"):
try:
chunk = await cast(Any, self.base_iterator).__anext__() # type: ignore[attr-defined]
# Capture the response ID from the first event to ensure consistency
if self._cached_response_id is None and hasattr(chunk, 'response'):
response_obj = getattr(chunk, 'response', None)
if response_obj and hasattr(response_obj, 'id'):
self._cached_response_id = response_obj.id
verbose_logger.debug(f"Cached response ID: {self._cached_response_id}")
# After emitting response.output_item.added, transition to MCP discovery
# Check if this is the output_item.added event
if not self.initial_events_emitted and hasattr(chunk, 'type'):
chunk_type = getattr(chunk, 'type', None)
if chunk_type == ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED:
self.initial_events_emitted = True
# Transition to MCP discovery phase after returning this chunk
self.phase = "mcp_discovery"
return chunk
# If auto-execution is enabled, check for completed responses
if self.should_auto_execute and self._is_response_completed(
chunk
):
# Collect the response for tool execution
response_obj = getattr(chunk, "response", None)
if isinstance(response_obj, ResponsesAPIResponse):
self.collected_response = response_obj
# Move to tool execution phase after emitting this chunk
self.phase = "tool_execution"
await self._generate_tool_execution_events()
return chunk
except StopAsyncIteration:
# Initial response ended, move to next phase
if self.should_auto_execute and self.collected_response:
self.phase = "tool_execution"
await self._generate_tool_execution_events()
else:
self.phase = "finished"
raise
else:
# base_iterator is not async iterable (likely a ResponsesAPIResponse)
# Collect it for tool execution if needed
if self.should_auto_execute and isinstance(
self.base_iterator, ResponsesAPIResponse
):
self.collected_response = self.base_iterator
self.phase = "tool_execution"
await self._generate_tool_execution_events()
else:
self.phase = "finished"
raise StopAsyncIteration
result = await self._handle_initial_response_phase()
if result is not None:
return result
# Phase 2: MCP Discovery Events (after response.output_item.added)
if self.phase == "mcp_discovery":
@ -523,6 +458,76 @@ class MCPEnhancedStreamingIterator(BaseResponsesAPIStreamingIterator):
# Should not reach here
raise StopAsyncIteration
async def _handle_initial_response_phase(
self,
) -> Optional[ResponsesAPIStreamingResponse]:
"""
Handle Phase 1: Initial Response Stream.
Returns a chunk to emit, or None to fall through to the next phase.
Raises StopAsyncIteration when the stream is exhausted with no auto-execution.
"""
if self.base_iterator is None:
await self._create_initial_response_iterator()
if self.base_iterator is None:
# LLM call failed — still emit MCP discovery events before finishing
if self.mcp_discovery_events:
self.phase = "mcp_discovery"
else:
self.phase = "finished"
raise StopAsyncIteration
return None
if self.base_iterator:
if hasattr(self.base_iterator, "__anext__"):
try:
chunk = await cast(Any, self.base_iterator).__anext__() # type: ignore[attr-defined]
# Capture the response ID from the first event to ensure consistency
if self._cached_response_id is None and hasattr(chunk, "response"):
response_obj = getattr(chunk, "response", None)
if response_obj and hasattr(response_obj, "id"):
self._cached_response_id = response_obj.id
verbose_logger.debug(f"Cached response ID: {self._cached_response_id}")
# After emitting response.output_item.added, transition to MCP discovery
if not self.initial_events_emitted and hasattr(chunk, "type"):
chunk_type = getattr(chunk, "type", None)
if chunk_type == ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED:
self.initial_events_emitted = True
self.phase = "mcp_discovery"
return chunk
# If auto-execution is enabled, check for completed responses
if self.should_auto_execute and self._is_response_completed(chunk):
response_obj = getattr(chunk, "response", None)
if isinstance(response_obj, ResponsesAPIResponse):
self.collected_response = response_obj
self.phase = "tool_execution"
await self._generate_tool_execution_events()
return chunk
except StopAsyncIteration:
if self.should_auto_execute and self.collected_response:
self.phase = "tool_execution"
await self._generate_tool_execution_events()
else:
self.phase = "finished"
raise
else:
# base_iterator is not async iterable (likely a ResponsesAPIResponse)
if self.should_auto_execute and isinstance(
self.base_iterator, ResponsesAPIResponse
):
self.collected_response = self.base_iterator
self.phase = "tool_execution"
await self._generate_tool_execution_events()
else:
self.phase = "finished"
raise StopAsyncIteration
return None
def _is_response_completed(self, chunk: ResponsesAPIStreamingResponse) -> bool:
"""Check if this chunk indicates the response is completed"""
from litellm.types.llms.openai import ResponsesAPIStreamEvents

View file

@ -363,6 +363,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
output_format: Optional[AnthropicOutputSchema] # Structured outputs support
speed: Optional[str] # Fast mode support for Opus models
output_config: Optional[AnthropicOutputConfig] # Configuration for Claude's output behavior
cache_control: Optional[Dict[str, Any]] # Automatic prompt caching
class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False):

View file

@ -3203,6 +3203,7 @@ class LlmProviders(str, Enum):
XIAOMI_MIMO = "xiaomi_mimo"
LITELLM_AGENT = "litellm_agent"
CURSOR = "cursor"
BEDROCK_MANTLE = "bedrock_mantle"
# Create a set of all provider values for quick lookup

View file

@ -4479,6 +4479,17 @@ def get_optional_params( # noqa: PLR0915
else False
),
)
elif custom_llm_provider == "bedrock_mantle":
optional_params = litellm.BedrockMantleChatConfig().map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=(
drop_params
if drop_params is not None and isinstance(drop_params, bool)
else False
),
)
elif custom_llm_provider == "deepseek":
optional_params = litellm.OpenAIConfig().map_openai_params(
non_default_params=non_default_params,
@ -7881,6 +7892,7 @@ class ProviderConfigManager:
# Simple provider mappings (no model parameter needed)
LlmProviders.DEEPSEEK: (lambda: litellm.DeepSeekChatConfig(), False),
LlmProviders.GROQ: (lambda: litellm.GroqChatConfig(), False),
LlmProviders.BEDROCK_MANTLE: (lambda: litellm.BedrockMantleChatConfig(), False),
LlmProviders.A2A: (lambda: litellm.A2AConfig(), False),
LlmProviders.BYTEZ: (lambda: litellm.BytezChatConfig(), False),
LlmProviders.DATABRICKS: (lambda: litellm.DatabricksConfig(), False),

View file

@ -5817,6 +5817,15 @@
],
"source": "https://devblogs.microsoft.com/foundry/whats-new-in-azure-ai-foundry-august-2025/#mistral-document-ai-(ocr)-%E2%80%94-serverless-in-foundry"
},
"azure_ai/mistral-document-ai-2512": {
"litellm_provider": "azure_ai",
"ocr_cost_per_page": 0.003,
"mode": "ocr",
"supported_endpoints": [
"/v1/ocr"
],
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/"
},
"azure_ai/doc-intelligence/prebuilt-read": {
"litellm_provider": "azure_ai",
"ocr_cost_per_page": 0.0015,
@ -39140,5 +39149,59 @@
"metadata": {
"notes": "DuckDuckGo Instant Answer API is free and does not require an API key."
}
},
"bedrock_mantle/openai.gpt-oss-120b": {
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 6e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"bedrock_mantle/openai.gpt-oss-20b": {
"input_cost_per_token": 7.5e-08,
"output_cost_per_token": 3e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"bedrock_mantle/openai.gpt-oss-safeguard-120b": {
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 6e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"bedrock_mantle/openai.gpt-oss-safeguard-20b": {
"input_cost_per_token": 7.5e-08,
"output_cost_per_token": 3e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
}
}

View file

@ -1,4 +1,3 @@
import json
import os
import sys
@ -13,7 +12,7 @@ from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
from litellm.types.utils import PromptTokensDetailsWrapper, ServerToolUse
from litellm.types.utils import ServerToolUse
def test_response_format_transformation_unit_test():
@ -1964,7 +1963,7 @@ def test_calculate_usage_completion_tokens_details_always_populated():
# completion_tokens_details should NOT be None
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.reasoning_tokens is 0
assert usage.completion_tokens_details.reasoning_tokens == 0
assert usage.completion_tokens_details.text_tokens == 248
assert usage.completion_tokens == 248
assert usage.prompt_tokens == 37
@ -2861,6 +2860,83 @@ def test_map_openai_params_with_context_management():
assert result["context_management"] == non_default_params_anthropic["context_management"]
def test_cache_control_in_supported_params():
"""
Test that cache_control is listed as a supported OpenAI param for Anthropic.
"""
config = AnthropicConfig()
params = config.get_supported_openai_params(model="claude-sonnet-4-20250514")
assert "cache_control" in params
def test_map_openai_params_with_cache_control():
"""
Test that map_openai_params correctly passes through top-level cache_control
for Anthropic's automatic prompt caching.
"""
config = AnthropicConfig()
non_default_params = {
"cache_control": {"type": "ephemeral"}
}
optional_params = {}
result = config.map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model="claude-sonnet-4-20250514",
drop_params=False,
)
assert "cache_control" in result
assert result["cache_control"] == {"type": "ephemeral"}
def test_map_openai_params_cache_control_ignored_when_not_dict():
"""
Test that cache_control is ignored when it is not a dict.
"""
config = AnthropicConfig()
non_default_params = {
"cache_control": "ephemeral"
}
optional_params = {}
result = config.map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model="claude-sonnet-4-20250514",
drop_params=False,
)
assert "cache_control" not in result
def test_transform_request_includes_cache_control():
"""
Test that transform_request includes top-level cache_control in the request body.
"""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
optional_params = {
"max_tokens": 100,
"cache_control": {"type": "ephemeral"},
}
result = config.transform_request(
model="claude-sonnet-4-20250514",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers={},
)
assert "cache_control" in result
assert result["cache_control"] == {"type": "ephemeral"}
def test_compaction_block_empty_list_not_added():
"""
Test that empty compaction_blocks list is not added to provider_specific_fields.
@ -2975,7 +3051,6 @@ def test_fast_mode_cost_calculation():
Test that fast mode applies the 'fast' multiplier from provider_specific_entry
on top of the base model cost (1.1x for claude-opus-4-6).
"""
from unittest.mock import MagicMock, patch
from litellm.llms.anthropic.cost_calculation import cost_per_token
from litellm.types.utils import Usage
@ -3015,7 +3090,6 @@ def test_fast_mode_with_inference_geo():
Test that fast mode + inference_geo both apply their multipliers from
provider_specific_entry (1.1 * 1.1 = 1.21x for claude-opus-4-6).
"""
from unittest.mock import patch
from litellm.llms.anthropic.cost_calculation import cost_per_token
from litellm.types.utils import Usage

View file

@ -239,6 +239,50 @@ class TestAzureAnthropicMessagesConfig:
assert "tools" in params
assert "tool_choice" in params
def test_transform_anthropic_messages_request_removes_scope_from_cache_control(
self,
):
"""Test that scope is removed from cache_control (Azure AI Foundry doesn't support it)"""
config = AzureAnthropicMessagesConfig()
model = "claude-sonnet-4-5"
messages = [
{
"role": "user",
"content": [
{
"type": "text",
"text": "Hello",
"cache_control": {"type": "ephemeral", "scope": "global"},
}
],
}
]
anthropic_messages_optional_request_params = {
"max_tokens": 1024,
"system": [
{
"type": "text",
"text": "You are an AI assistant.",
"cache_control": {"type": "ephemeral", "scope": "global"},
}
],
}
litellm_params = GenericLiteLLMParams()
headers = {}
result = config.transform_anthropic_messages_request(
model=model,
messages=messages,
anthropic_messages_optional_request_params=anthropic_messages_optional_request_params,
litellm_params=litellm_params,
headers=headers,
)
assert "scope" not in result["system"][0]["cache_control"]
assert result["system"][0]["cache_control"]["type"] == "ephemeral"
assert "scope" not in result["messages"][0]["content"][0]["cache_control"]
assert result["messages"][0]["content"][0]["cache_control"]["type"] == "ephemeral"
class TestProviderConfigManagerAzureAnthropicMessages:
"""Test ProviderConfigManager returns correct config for Azure AI Anthropic Messages API"""

View file

@ -178,3 +178,48 @@ def test_remove_ttl_from_cache_control():
request5 = {}
cfg._remove_ttl_from_cache_control(request5)
assert request5 == {}
def test_remove_scope_from_cache_control():
"""Ensure scope field is removed from cache_control for Bedrock (not supported)."""
cfg = AmazonAnthropicClaudeMessagesConfig()
# Test case 1: System with cache_control containing scope
request = {
"system": [
{
"type": "text",
"text": "You are an AI assistant.",
"cache_control": {
"type": "ephemeral",
"scope": "global",
},
}
],
"messages": [
{
"role": "user",
"content": [
{
"type": "text",
"text": "Hello",
"cache_control": {
"type": "ephemeral",
"scope": "global",
},
}
],
}
],
}
cfg._remove_ttl_from_cache_control(request)
# Verify scope is removed from system
assert "scope" not in request["system"][0]["cache_control"]
assert request["system"][0]["cache_control"]["type"] == "ephemeral"
# Verify scope is removed from messages
assert "scope" not in request["messages"][0]["content"][0]["cache_control"]
assert request["messages"][0]["content"][0]["cache_control"]["type"] == "ephemeral"

View file

@ -0,0 +1,169 @@
"""
Unit tests for Amazon Bedrock Mantle provider configuration.
Bedrock Mantle is Amazon Bedrock's OpenAI-compatible inference engine (Project Mantle).
API docs: https://docs.aws.amazon.com/bedrock/latest/userguide/bedrock-mantle.html
"""
import os
import sys
sys.path.insert(0, os.path.abspath("../../../../.."))
import pytest
import litellm
from litellm.llms.bedrock_mantle.chat.transformation import BedrockMantleChatConfig
from litellm.types.utils import LlmProviders
class TestBedrockMantleProviderRegistration:
def test_provider_enum_exists(self):
assert LlmProviders.BEDROCK_MANTLE == "bedrock_mantle"
def test_provider_in_provider_list(self):
assert "bedrock_mantle" in litellm.provider_list
def test_models_loaded(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
litellm.add_known_models()
assert len(litellm.bedrock_mantle_models) > 0
assert "bedrock_mantle/openai.gpt-oss-120b" in litellm.bedrock_mantle_models
assert "bedrock_mantle/openai.gpt-oss-20b" in litellm.bedrock_mantle_models
assert (
"bedrock_mantle/openai.gpt-oss-safeguard-120b" in litellm.bedrock_mantle_models
)
assert (
"bedrock_mantle/openai.gpt-oss-safeguard-20b" in litellm.bedrock_mantle_models
)
class TestBedrockMantleConfig:
def test_custom_llm_provider(self):
cfg = BedrockMantleChatConfig()
assert cfg.custom_llm_provider == "bedrock_mantle"
def test_default_api_base_uses_env_region(self, monkeypatch):
monkeypatch.setenv("BEDROCK_MANTLE_REGION", "eu-west-1")
monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False)
cfg = BedrockMantleChatConfig()
api_base, _ = cfg._get_openai_compatible_provider_info(None, None)
assert api_base == "https://bedrock-mantle.eu-west-1.api.aws/v1"
def test_default_api_base_uses_aws_region(self, monkeypatch):
monkeypatch.delenv("BEDROCK_MANTLE_REGION", raising=False)
monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False)
monkeypatch.setenv("AWS_REGION", "ap-northeast-1")
cfg = BedrockMantleChatConfig()
api_base, _ = cfg._get_openai_compatible_provider_info(None, None)
assert api_base == "https://bedrock-mantle.ap-northeast-1.api.aws/v1"
def test_default_api_base_fallback_to_us_east_1(self, monkeypatch):
monkeypatch.delenv("BEDROCK_MANTLE_REGION", raising=False)
monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False)
monkeypatch.delenv("AWS_REGION", raising=False)
cfg = BedrockMantleChatConfig()
api_base, _ = cfg._get_openai_compatible_provider_info(None, None)
assert api_base == "https://bedrock-mantle.us-east-1.api.aws/v1"
def test_custom_api_base_overrides_default(self, monkeypatch):
custom_base = "https://bedrock-mantle.us-west-2.api.aws/v1"
cfg = BedrockMantleChatConfig()
api_base, _ = cfg._get_openai_compatible_provider_info(custom_base, None)
assert api_base == custom_base
def test_api_key_from_env(self, monkeypatch):
monkeypatch.setenv("BEDROCK_MANTLE_API_KEY", "test-key-123")
cfg = BedrockMantleChatConfig()
_, api_key = cfg._get_openai_compatible_provider_info(None, None)
assert api_key == "test-key-123"
def test_api_key_param_overrides_env(self, monkeypatch):
monkeypatch.setenv("BEDROCK_MANTLE_API_KEY", "env-key")
cfg = BedrockMantleChatConfig()
_, api_key = cfg._get_openai_compatible_provider_info(None, "explicit-key")
assert api_key == "explicit-key"
def test_get_supported_openai_params(self):
cfg = BedrockMantleChatConfig()
params = cfg.get_supported_openai_params("openai.gpt-oss-120b")
assert "tools" in params
assert "tool_choice" in params
assert "temperature" in params
assert "stream" in params
assert "max_tokens" in params
class TestBedrockMantleProviderResolution:
def test_get_llm_provider_resolves_correctly(self):
model, provider, _, _ = litellm.get_llm_provider(
"bedrock_mantle/openai.gpt-oss-120b"
)
assert provider == "bedrock_mantle"
assert model == "openai.gpt-oss-120b"
def test_get_llm_provider_20b(self):
model, provider, _, _ = litellm.get_llm_provider(
"bedrock_mantle/openai.gpt-oss-20b"
)
assert provider == "bedrock_mantle"
assert model == "openai.gpt-oss-20b"
class TestBedrockMantlePricing:
"""Tests that verify Bedrock Mantle uses correct AWS Bedrock pricing, not OpenAI pricing."""
def test_gpt_oss_120b_pricing(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
litellm.add_known_models()
info = litellm.get_model_info("bedrock_mantle/openai.gpt-oss-120b")
# Bedrock pricing: $0.15/M input, $0.60/M output
assert info["input_cost_per_token"] == pytest.approx(1.5e-7)
assert info["output_cost_per_token"] == pytest.approx(6e-7)
def test_gpt_oss_20b_pricing(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
litellm.add_known_models()
info = litellm.get_model_info("bedrock_mantle/openai.gpt-oss-20b")
# Bedrock pricing: $0.075/M input, $0.30/M output
assert info["input_cost_per_token"] == pytest.approx(7.5e-8)
assert info["output_cost_per_token"] == pytest.approx(3e-7)
def test_pricing_significantly_cheaper_than_openai_native(self, monkeypatch):
"""
Verify Bedrock Mantle pricing is cheaper than OpenAI's direct API pricing.
This is the core issue the provider addition fixes — previously users were being
billed at OpenAI rates instead of the cheaper Bedrock rates.
"""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
litellm.add_known_models()
bedrock_info = litellm.get_model_info("bedrock_mantle/openai.gpt-oss-120b")
# OpenAI direct pricing for gpt-oss-120b is ~$0.039/M input, $0.190/M output
# Bedrock should be cheaper at $0.15/M input and $0.60/M output... wait
# Actually, Bedrock ADDS value not reduces cost vs OpenAI direct for these models.
# The key fix is that we now use Bedrock-specific prices instead of mapping to
# some unrelated OpenAI model (like gpt-4) pricing.
# Just validate the pricing is as expected from AWS docs.
assert bedrock_info["input_cost_per_token"] == pytest.approx(1.5e-7)
assert bedrock_info["output_cost_per_token"] == pytest.approx(6e-7)
def test_safeguard_models_have_larger_output_tokens(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
litellm.add_known_models()
info_120b = litellm.get_model_info("bedrock_mantle/openai.gpt-oss-120b")
info_safeguard = litellm.get_model_info(
"bedrock_mantle/openai.gpt-oss-safeguard-120b"
)
assert info_safeguard["max_output_tokens"] > info_120b["max_output_tokens"]
def test_reasoning_support(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
litellm.add_known_models()
info = litellm.get_model_info("bedrock_mantle/openai.gpt-oss-120b")
assert info.get("supports_reasoning") is True
def test_context_window(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
litellm.add_known_models()
info = litellm.get_model_info("bedrock_mantle/openai.gpt-oss-120b")
assert info["max_input_tokens"] == 131072

View file

@ -489,3 +489,112 @@ def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_hea
assert (
"anthropic-beta" not in headers2
), "Header should be removed if no supported values remain"
def test_vertex_ai_anthropic_output_config_dropped():
"""
Test that output_config parameter is dropped from Vertex AI Anthropic requests.
Vertex AI does not support the output_config parameter (used for effort settings
in Anthropic API). This test ensures it's properly removed to prevent
"Extra inputs are not permitted" errors.
"""
config = VertexAIAnthropicConfig()
messages = [{"role": "user", "content": "What is 2+2?"}]
headers = {}
# Simulate optional_params with output_config that would be passed in
optional_params = {
"max_tokens": 1024,
"output_config": {
"effort": "high" # This is Anthropic-specific and not supported by Vertex AI
},
}
# Call transform_request which should drop output_config
result = config.transform_request(
model="claude-3-5-sonnet-20241022",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers=headers,
)
# Verify output_config was removed
assert "output_config" not in result, \
"output_config should be dropped from Vertex AI Anthropic requests"
# Verify other parameters are preserved
assert result["max_tokens"] == 1024, "max_tokens should be preserved"
assert "messages" in result, "messages should be present"
def test_vertex_ai_anthropic_output_format_and_output_config_both_dropped():
"""
Test that both output_format and output_config are dropped from Vertex AI requests.
This ensures that even if both parameters somehow make it to the transform_request,
they are properly cleaned up before sending to Vertex AI.
"""
config = VertexAIAnthropicConfig()
messages = [{"role": "user", "content": "Extract structured data"}]
headers = {}
optional_params = {
"max_tokens": 2048,
"output_format": {
"type": "json_schema",
"json_schema": {
"name": "data",
"schema": {"type": "object", "properties": {"result": {"type": "string"}}}
}
},
"output_config": {
"effort": "high"
},
}
# Simulate parent class creating test_data with both parameters
# (as if the parent transform_request added them)
test_data = {
"model": "claude-3-5-sonnet-20241022",
"messages": messages,
"max_tokens": 2048,
"output_format": optional_params["output_format"],
"output_config": optional_params["output_config"],
}
# Mock the parent transform_request to return data with both parameters
original_transform = config.__class__.__bases__[0].transform_request
def mock_transform_request(self, model, messages, optional_params, litellm_params, headers):
return test_data.copy()
config.__class__.__bases__[0].transform_request = mock_transform_request
try:
result = config.transform_request(
model="claude-3-5-sonnet-20241022",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers=headers,
)
# Verify both were removed
assert "output_format" not in result, \
"output_format should be dropped from Vertex AI requests"
assert "output_config" not in result, \
"output_config should be dropped from Vertex AI requests"
# Verify essential params are preserved
assert result["max_tokens"] == 2048, "max_tokens should be preserved"
assert "messages" in result, "messages should be present"
assert "model" not in result, "model should also be dropped for Vertex AI"
finally:
# Restore original method
config.__class__.__bases__[0].transform_request = original_transform

View file

@ -677,6 +677,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"video_generation",
"moderation",
"rerank",
"realtime",
"responses",
"ocr",
"search",

View file

@ -9,6 +9,7 @@ export enum Providers {
AssemblyAI = "AssemblyAI",
AUTO_ROUTER = "Auto Router",
Bedrock = "Amazon Bedrock",
BedrockMantle = "Amazon Bedrock Mantle",
SageMaker = "AWS SageMaker",
Azure = "Azure",
Azure_AI_Studio = "Azure AI Foundry (Studio)",
@ -119,6 +120,7 @@ export const provider_map: Record<string, string> = {
AZURE_TEXT: "azure_text",
BASETEN: "baseten",
Bedrock: "bedrock",
BedrockMantle: "bedrock_mantle",
BYTEZ: "bytez",
Cerebras: "cerebras",
CLARIFAI: "clarifai",
@ -226,6 +228,7 @@ export const providerLogoMap: Record<string, string> = {
[Providers.AZURE_TEXT]: `${asset_logos_folder}microsoft_azure.svg`,
[Providers.BASETEN]: `${asset_logos_folder}baseten.svg`,
[Providers.Bedrock]: `${asset_logos_folder}bedrock.svg`,
[Providers.BedrockMantle]: `${asset_logos_folder}bedrock.svg`,
[Providers.SageMaker]: `${asset_logos_folder}bedrock.svg`,
[Providers.Cerebras]: `${asset_logos_folder}cerebras.svg`,
[Providers.CLOUDFLARE]: `${asset_logos_folder}cloudflare.svg`,