mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-30 01:52:18 +00:00
feat(rust): shape Anthropic Messages requests natively
The Rust Messages route only relayed the body. It now runs the request shaping the Python handler does for the direct Anthropic provider: history sanitizers (empty blocks, tool ids, replayed web search results, provider_specific_fields, encrypted reasoning, advisor blocks), reasoning_effort and adaptive/legacy thinking translation against the model's capability flags, the sampling and speed gates under drop_params, the metadata allowlist, additional_drop_params, reasoning auto summary, OAuth and ANTHROPIC_AUTH_TOKEN credentials, provider_specific_header merging and anthropic-beta injection. Capability flags and LiteLLM settings reach Rust through route_host.shaping(). A request the route rejects before the call now maps to BadRequestError instead of APIConnectionError Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
ae00e0f37c
commit
033cafaf28
22 changed files with 3241 additions and 92 deletions
110
litellm-rust/crates/core-utils/src/dot_notation_indexing.rs
Normal file
110
litellm-rust/crates/core-utils/src/dot_notation_indexing.rs
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
//! JSONPath-like field deletion for `additional_drop_params`: `field`, `parent.child`,
|
||||
//! `array[*].field` and `array[0].field`, as Python's `delete_nested_value` reads them.
|
||||
|
||||
use serde_json::Value;
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
enum Segment {
|
||||
Field(String),
|
||||
Every,
|
||||
Index(usize),
|
||||
}
|
||||
|
||||
fn parse_segments(path: &str) -> Option<Vec<Segment>> {
|
||||
let mut segments = Vec::new();
|
||||
let mut rest = path;
|
||||
while !rest.is_empty() {
|
||||
if let Some(after_open) = rest.strip_prefix('[') {
|
||||
let (inside, after) = after_open.split_once(']')?;
|
||||
segments.push(match inside {
|
||||
"*" => Segment::Every,
|
||||
index => Segment::Index(index.trim().parse().ok()?),
|
||||
});
|
||||
rest = after.strip_prefix('.').unwrap_or(after);
|
||||
continue;
|
||||
}
|
||||
let end = rest.find(['.', '[']).unwrap_or(rest.len());
|
||||
let (field, after) = rest.split_at(end);
|
||||
if !field.is_empty() {
|
||||
segments.push(Segment::Field(field.to_string()));
|
||||
}
|
||||
rest = after.strip_prefix('.').unwrap_or(after);
|
||||
}
|
||||
Some(segments)
|
||||
}
|
||||
|
||||
fn without_path(value: Value, segments: &[Segment]) -> Value {
|
||||
let Some((segment, tail)) = segments.split_first() else {
|
||||
return value;
|
||||
};
|
||||
match (segment, value) {
|
||||
(Segment::Field(name), Value::Object(object)) => Value::Object(
|
||||
object
|
||||
.into_iter()
|
||||
.filter_map(|(key, item)| {
|
||||
if key != *name {
|
||||
return Some((key, item));
|
||||
}
|
||||
(!tail.is_empty()).then(|| (key, without_path(item, tail)))
|
||||
})
|
||||
.collect(),
|
||||
),
|
||||
(Segment::Every, Value::Array(items)) if !tail.is_empty() => Value::Array(
|
||||
items
|
||||
.into_iter()
|
||||
.map(|item| without_path(item, tail))
|
||||
.collect(),
|
||||
),
|
||||
(Segment::Index(index), Value::Array(items)) if !tail.is_empty() => Value::Array(
|
||||
items
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
.map(|(position, item)| {
|
||||
if position == *index {
|
||||
without_path(item, tail)
|
||||
} else {
|
||||
item
|
||||
}
|
||||
})
|
||||
.collect(),
|
||||
),
|
||||
(_, value) => value,
|
||||
}
|
||||
}
|
||||
|
||||
/// The value with the field at `path` removed. An unparsable path leaves it untouched, and
|
||||
/// array elements themselves are never removed, only fields inside them.
|
||||
pub fn delete_nested_value(value: Value, path: &str) -> Value {
|
||||
match parse_segments(path) {
|
||||
Some(segments) => without_path(value, &segments),
|
||||
None => value,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use rstest::rstest;
|
||||
use serde_json::json;
|
||||
|
||||
use super::*;
|
||||
|
||||
const TOOLS: &str = r#"[{"name": "t1", "input_examples": ["a"]}, {"name": "t2", "keep": 1, "input_examples": ["b"]}]"#;
|
||||
|
||||
fn body() -> Value {
|
||||
let tools: Value = serde_json::from_str(TOOLS).unwrap();
|
||||
json!({"tools": tools, "metadata": {"user_id": "u"}})
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case("tools[*].input_examples", json!({"tools": [{"name": "t1"}, {"name": "t2", "keep": 1}], "metadata": {"user_id": "u"}}))]
|
||||
#[case("tools[0].input_examples", json!({"tools": [{"name": "t1"}, {"name": "t2", "keep": 1, "input_examples": ["b"]}], "metadata": {"user_id": "u"}}))]
|
||||
#[case("metadata.user_id", json!({"tools": body()["tools"], "metadata": {}}))]
|
||||
#[case("tools", json!({"metadata": {"user_id": "u"}}))]
|
||||
#[case("tools[*]", body())]
|
||||
#[case("missing.path", body())]
|
||||
#[case("tools[x].name", body())]
|
||||
#[case("tools[1", body())]
|
||||
fn deletes_like_python(#[case] path: &str, #[case] expected: Value) {
|
||||
assert_eq!(delete_nested_value(body(), path), expected);
|
||||
}
|
||||
}
|
||||
|
|
@ -1,5 +1,6 @@
|
|||
pub mod call_arguments;
|
||||
pub mod core_helpers;
|
||||
pub mod dot_notation_indexing;
|
||||
pub mod exception_mapping_utils;
|
||||
pub mod get_llm_provider_logic;
|
||||
pub mod params;
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
use litellm_http::request::string_headers as shared_string_headers;
|
||||
pub(super) use litellm_http::request::{has_bearer_auth, has_header, truncate_error_body};
|
||||
pub(super) use litellm_http::request::truncate_error_body;
|
||||
use litellm_llms::{
|
||||
anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG,
|
||||
azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG,
|
||||
|
|
|
|||
|
|
@ -32,6 +32,7 @@ pub async fn messages(request: MessagesRequest<'_>) -> Result<AnthropicMessagesR
|
|||
custom_llm_provider: request.custom_llm_provider.map(Into::into),
|
||||
extra_headers: request.extra_headers,
|
||||
timeout: request.timeout,
|
||||
shaping: request.shaping,
|
||||
};
|
||||
match litellm_host::run::run(messages_machine(), &LocalMessagesHost::new(call)).await? {
|
||||
MessagesOutput::Message(message) => Ok(*message),
|
||||
|
|
|
|||
|
|
@ -1,15 +1,24 @@
|
|||
use litellm_core_utils::get_llm_provider_logic::{CustomLlmProvider, get_custom_llm_provider};
|
||||
use litellm_llms::base_llm::anthropic_messages::transformation::{
|
||||
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
|
||||
use litellm_core_utils::{
|
||||
dot_notation_indexing::delete_nested_value,
|
||||
get_llm_provider_logic::{CustomLlmProvider, get_custom_llm_provider},
|
||||
};
|
||||
use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest;
|
||||
use serde_json::{Map, Value};
|
||||
use litellm_llms::{
|
||||
anthropic::common_utils::{
|
||||
flatten_unencrypted_web_search_results, sanitize_tool_use_ids, strip_empty_content_blocks,
|
||||
strip_provider_specific_fields,
|
||||
},
|
||||
base_llm::anthropic_messages::transformation::MessagesTransformContext,
|
||||
};
|
||||
use litellm_types::llms::anthropic_messages::anthropic_request::{
|
||||
AnthropicMessage, AnthropicMessagesRequest,
|
||||
};
|
||||
use serde_json::{Map, Value, json};
|
||||
|
||||
use super::{
|
||||
Error,
|
||||
common_utils::{has_bearer_auth, has_header, messages_provider_config, string_headers},
|
||||
common_utils::{messages_provider_config, string_headers},
|
||||
};
|
||||
use crate::messages::types::{MessagesRequest, ProviderMessagesRequest};
|
||||
use crate::messages::types::{MessagesRequest, MessagesShaping, ProviderMessagesRequest};
|
||||
|
||||
pub(super) fn prepare_provider_request(
|
||||
request: MessagesRequest<'_>,
|
||||
|
|
@ -35,17 +44,33 @@ pub(super) fn prepare_provider_request(
|
|||
.ok_or_else(|| Error::InvalidProvider(provider.to_string()))?;
|
||||
let env_lookup = |key: &str| std::env::var(key).ok();
|
||||
|
||||
let headers =
|
||||
validate_environment(config, request.extra_headers, request.api_key, &env_lookup)?;
|
||||
|
||||
let typed_request: AnthropicMessagesRequest =
|
||||
serde_json::from_value(request.body).map_err(|err| {
|
||||
Error::InvalidRequest(format!("invalid Anthropic messages request: {err}"))
|
||||
})?;
|
||||
let transformed = config.transform_anthropic_messages_request(AnthropicMessagesRequest {
|
||||
model: model.clone(),
|
||||
..typed_request
|
||||
let body = request
|
||||
.shaping
|
||||
.additional_drop_params
|
||||
.iter()
|
||||
.fold(request.body, |body, path| delete_nested_value(body, path));
|
||||
let typed_request: AnthropicMessagesRequest = serde_json::from_value(body).map_err(|err| {
|
||||
Error::InvalidRequest(format!("invalid Anthropic messages request: {err}"))
|
||||
})?;
|
||||
let sanitized = sanitize_request(
|
||||
AnthropicMessagesRequest {
|
||||
model: model.clone(),
|
||||
..typed_request
|
||||
},
|
||||
&request.shaping,
|
||||
);
|
||||
let transformed = config.transform_anthropic_messages_request(
|
||||
sanitized,
|
||||
&MessagesTransformContext::new(request.shaping.capabilities, request.shaping.drop_params),
|
||||
)?;
|
||||
|
||||
let forwarded = string_headers(request.extra_headers)?;
|
||||
let authenticated = config.authenticate(forwarded, request.api_key, &env_lookup)?;
|
||||
let headers = config.request_headers(
|
||||
with_default_headers(authenticated, config.default_headers()),
|
||||
&transformed,
|
||||
);
|
||||
|
||||
let body = serde_json::to_value(transformed).map_err(|err| {
|
||||
Error::InvalidRequest(format!(
|
||||
"failed to serialize Anthropic messages request: {err}"
|
||||
|
|
@ -65,33 +90,66 @@ pub(super) fn prepare_provider_request(
|
|||
})
|
||||
}
|
||||
|
||||
fn validate_environment(
|
||||
config: &dyn BaseAnthropicMessagesConfig,
|
||||
extra_headers: Option<Map<String, Value>>,
|
||||
api_key: Option<&str>,
|
||||
env_lookup: &dyn Fn(&str) -> Option<String>,
|
||||
) -> Result<Vec<(String, String)>, Error> {
|
||||
let mut headers = string_headers(extra_headers)?;
|
||||
|
||||
let auth_strategy = config.auth_strategy();
|
||||
let already_authorized = has_header(&headers, auth_strategy.header_name())
|
||||
|| (config.accepts_bearer_auth() && has_bearer_auth(&headers));
|
||||
if !already_authorized {
|
||||
let api_key = config.resolve_api_key(api_key, env_lookup)?;
|
||||
let auth_header = match auth_strategy {
|
||||
MessagesAuthStrategy::Bearer => {
|
||||
("authorization".to_string(), format!("Bearer {api_key}"))
|
||||
}
|
||||
MessagesAuthStrategy::Header(name) => (name.to_string(), api_key),
|
||||
};
|
||||
headers.push(auth_header);
|
||||
/// The route-level cleanup Python's `anthropic_messages` runs before any provider config:
|
||||
/// history sanitizers, the `metadata` allowlist and the reasoning auto summary.
|
||||
fn sanitize_request(
|
||||
request: AnthropicMessagesRequest,
|
||||
shaping: &MessagesShaping,
|
||||
) -> AnthropicMessagesRequest {
|
||||
AnthropicMessagesRequest {
|
||||
messages: sanitize_messages(request.messages),
|
||||
metadata: request.metadata.as_ref().map(allowed_metadata),
|
||||
thinking: with_reasoning_auto_summary(request.thinking, shaping.reasoning_auto_summary),
|
||||
..request
|
||||
}
|
||||
|
||||
for (name, value) in config.default_headers() {
|
||||
if !has_header(&headers, name) {
|
||||
headers.push((name.to_string(), value.to_string()));
|
||||
}
|
||||
}
|
||||
|
||||
Ok(headers)
|
||||
}
|
||||
|
||||
fn sanitize_messages(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
|
||||
strip_provider_specific_fields(flatten_unencrypted_web_search_results(
|
||||
sanitize_tool_use_ids(strip_empty_content_blocks(messages)),
|
||||
))
|
||||
}
|
||||
|
||||
/// Only the fields Anthropic's `metadata` accepts reach the provider; LiteLLM-specific
|
||||
/// metadata travels under `litellm_metadata` instead.
|
||||
fn allowed_metadata(metadata: &Value) -> Value {
|
||||
let user_id = metadata.get("user_id").filter(|value| !value.is_null());
|
||||
Value::Object(
|
||||
user_id
|
||||
.map(|value| ("user_id".to_string(), value.clone()))
|
||||
.into_iter()
|
||||
.collect::<Map<String, Value>>(),
|
||||
)
|
||||
}
|
||||
|
||||
fn with_reasoning_auto_summary(thinking: Option<Value>, enabled: bool) -> Option<Value> {
|
||||
let Some(Value::Object(thinking)) = thinking else {
|
||||
return thinking;
|
||||
};
|
||||
if !enabled || thinking.get("type").and_then(Value::as_str) == Some("disabled") {
|
||||
return Some(Value::Object(thinking));
|
||||
}
|
||||
Some(Value::Object(
|
||||
thinking
|
||||
.into_iter()
|
||||
.filter(|(key, _)| key != "display")
|
||||
.chain([("display".to_string(), json!("summarized"))])
|
||||
.collect(),
|
||||
))
|
||||
}
|
||||
|
||||
fn with_default_headers(
|
||||
headers: Vec<(String, String)>,
|
||||
defaults: &[(&str, &str)],
|
||||
) -> Vec<(String, String)> {
|
||||
let missing: Vec<(String, String)> = defaults
|
||||
.iter()
|
||||
.filter(|(name, _)| {
|
||||
!headers
|
||||
.iter()
|
||||
.any(|(header, _)| header.eq_ignore_ascii_case(name))
|
||||
})
|
||||
.map(|(name, value)| ((*name).to_string(), (*value).to_string()))
|
||||
.collect();
|
||||
headers.into_iter().chain(missing).collect()
|
||||
}
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ use super::{
|
|||
common_utils::messages_provider_config,
|
||||
handler::{decode_response, network, provider_error, send},
|
||||
prepare::prepare_provider_request,
|
||||
types::MessagesRequest,
|
||||
types::{MessagesRequest, MessagesShaping},
|
||||
};
|
||||
use crate::constants::ANTHROPIC_MESSAGES_PROVIDER;
|
||||
|
||||
|
|
@ -39,6 +39,7 @@ pub struct MessagesCall {
|
|||
pub custom_llm_provider: Option<String>,
|
||||
pub extra_headers: Option<Map<String, Value>>,
|
||||
pub timeout: Option<Duration>,
|
||||
pub shaping: MessagesShaping,
|
||||
}
|
||||
|
||||
impl MessagesCall {
|
||||
|
|
@ -135,6 +136,7 @@ async fn execute(host: MessagesHost) -> Result<MessagesOutput, Error> {
|
|||
custom_llm_provider: call.custom_llm_provider.as_deref(),
|
||||
extra_headers: call.extra_headers.clone(),
|
||||
timeout: call.timeout,
|
||||
shaping: call.shaping.clone(),
|
||||
})?;
|
||||
if stream && request.provider != ANTHROPIC_MESSAGES_PROVIDER {
|
||||
return Err(Error::Unsupported("streaming messages for this provider"));
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
use std::time::Duration;
|
||||
|
||||
use litellm_http::request::{has_bearer_auth, has_header};
|
||||
use serde_json::{Map, Value, json};
|
||||
use tokio::{
|
||||
io::{AsyncReadExt, AsyncWriteExt},
|
||||
|
|
@ -8,12 +9,10 @@ use tokio::{
|
|||
|
||||
use super::{
|
||||
Error,
|
||||
common_utils::{
|
||||
has_bearer_auth, has_header, messages_provider_config, string_headers, truncate_error_body,
|
||||
},
|
||||
common_utils::{messages_provider_config, string_headers, truncate_error_body},
|
||||
messages,
|
||||
};
|
||||
use crate::messages::types::MessagesRequest;
|
||||
use crate::messages::types::{MessagesRequest, MessagesShaping};
|
||||
|
||||
async fn read_http_request(socket: &mut TcpStream) -> String {
|
||||
let mut request = Vec::new();
|
||||
|
|
@ -160,6 +159,7 @@ async fn messages_round_trip_builds_azure_request_and_passes_response_through()
|
|||
custom_llm_provider: Some("azure_ai"),
|
||||
extra_headers: None,
|
||||
timeout: Some(Duration::from_secs(5)),
|
||||
shaping: MessagesShaping::default(),
|
||||
})
|
||||
.await
|
||||
.expect("messages request succeeds");
|
||||
|
|
@ -216,6 +216,7 @@ async fn messages_round_trip_builds_native_anthropic_request() {
|
|||
custom_llm_provider: Some("anthropic"),
|
||||
extra_headers: None,
|
||||
timeout: Some(Duration::from_secs(5)),
|
||||
shaping: MessagesShaping::default(),
|
||||
})
|
||||
.await
|
||||
.expect("messages request succeeds");
|
||||
|
|
@ -269,6 +270,7 @@ async fn messages_does_not_duplicate_auth_when_x_api_key_supplied() {
|
|||
custom_llm_provider: Some("azure_ai"),
|
||||
extra_headers: Some(headers),
|
||||
timeout: Some(Duration::from_secs(5)),
|
||||
shaping: MessagesShaping::default(),
|
||||
})
|
||||
.await
|
||||
.expect("messages request succeeds");
|
||||
|
|
@ -323,6 +325,7 @@ async fn messages_forwards_entra_id_bearer_without_requiring_api_key() {
|
|||
custom_llm_provider: Some("azure_ai"),
|
||||
extra_headers: Some(headers),
|
||||
timeout: Some(Duration::from_secs(5)),
|
||||
shaping: MessagesShaping::default(),
|
||||
})
|
||||
.await
|
||||
.expect("entra id request succeeds without api key");
|
||||
|
|
@ -347,6 +350,7 @@ async fn messages_requires_auth_when_no_key_and_no_header() {
|
|||
custom_llm_provider: Some("azure_ai"),
|
||||
extra_headers: None,
|
||||
timeout: Some(Duration::from_millis(50)),
|
||||
shaping: MessagesShaping::default(),
|
||||
})
|
||||
.await
|
||||
.expect_err("missing auth errors");
|
||||
|
|
@ -385,6 +389,7 @@ async fn messages_ignores_malformed_authorization_and_uses_api_key() {
|
|||
custom_llm_provider: Some("azure_ai"),
|
||||
extra_headers: Some(headers),
|
||||
timeout: Some(Duration::from_secs(5)),
|
||||
shaping: MessagesShaping::default(),
|
||||
})
|
||||
.await
|
||||
.expect("falls back to api key");
|
||||
|
|
@ -426,6 +431,7 @@ async fn messages_maps_provider_error_status_to_http_error() {
|
|||
custom_llm_provider: Some("azure_ai"),
|
||||
extra_headers: None,
|
||||
timeout: Some(Duration::from_secs(5)),
|
||||
shaping: MessagesShaping::default(),
|
||||
})
|
||||
.await
|
||||
.expect_err("provider error propagates");
|
||||
|
|
@ -446,6 +452,7 @@ async fn messages_rejects_unsupported_provider() {
|
|||
custom_llm_provider: Some("openai"),
|
||||
extra_headers: None,
|
||||
timeout: Some(Duration::from_millis(50)),
|
||||
shaping: MessagesShaping::default(),
|
||||
})
|
||||
.await
|
||||
.expect_err("unsupported provider errors");
|
||||
|
|
|
|||
|
|
@ -1,8 +1,29 @@
|
|||
use std::time::Duration;
|
||||
|
||||
use litellm_llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
|
||||
use litellm_llms::{
|
||||
anthropic::common_utils::AnthropicModelCapabilities,
|
||||
base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
/// What the host knows about the deployment that the route cannot read off the body: the
|
||||
/// model's cost-map flags and the caller's LiteLLM-level request settings.
|
||||
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
|
||||
pub struct MessagesShaping {
|
||||
#[serde(default)]
|
||||
pub capabilities: AnthropicModelCapabilities,
|
||||
/// `litellm.drop_params` or the per-request `drop_params`.
|
||||
#[serde(default)]
|
||||
pub drop_params: bool,
|
||||
/// `litellm.reasoning_auto_summary` or `LITELLM_REASONING_AUTO_SUMMARY`.
|
||||
#[serde(default)]
|
||||
pub reasoning_auto_summary: bool,
|
||||
/// The deployment's `additional_drop_params`: dot paths removed from the request body.
|
||||
#[serde(default)]
|
||||
pub additional_drop_params: Vec<String>,
|
||||
}
|
||||
|
||||
pub struct MessagesRequest<'a> {
|
||||
pub model: &'a str,
|
||||
pub body: Value,
|
||||
|
|
@ -11,6 +32,7 @@ pub struct MessagesRequest<'a> {
|
|||
pub custom_llm_provider: Option<&'a str>,
|
||||
pub extra_headers: Option<Map<String, Value>>,
|
||||
pub timeout: Option<Duration>,
|
||||
pub shaping: MessagesShaping,
|
||||
}
|
||||
|
||||
pub struct ProviderMessagesRequest {
|
||||
|
|
|
|||
827
litellm-rust/crates/llms/src/anthropic/common_utils.rs
Normal file
827
litellm-rust/crates/llms/src/anthropic/common_utils.rs
Normal file
|
|
@ -0,0 +1,827 @@
|
|||
//! Anthropic knowledge shared by every route that speaks the Messages API: model
|
||||
//! capability flags, beta header values, and the message sanitizers Python runs in
|
||||
//! `litellm/llms/anthropic/common_utils.py`.
|
||||
|
||||
use litellm_types::llms::anthropic_messages::anthropic_request::{
|
||||
AnthropicMessage, ContentBlock, MessageContent,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use serde_json::Value;
|
||||
|
||||
use crate::anthropic::ANTHROPIC_OAUTH_TOKEN_PREFIX;
|
||||
|
||||
pub const ANTHROPIC_OAUTH_BETA_HEADER: &str = "oauth-2025-04-20";
|
||||
pub const ANTHROPIC_ADVISOR_TOOL_TYPE: &str = "advisor_20260301";
|
||||
pub const ANTHROPIC_TOOL_SEARCH_TOOL_TYPES: [&str; 2] = [
|
||||
"tool_search_tool_regex_20251119",
|
||||
"tool_search_tool_bm25_20251119",
|
||||
];
|
||||
pub const ENCRYPTED_REASONING_SIGNATURE_PREFIX: &str = "litellm_encrypted_reasoning:";
|
||||
const THOUGHT_SIGNATURE_SEPARATOR: &str = "__thought__";
|
||||
|
||||
/// Known `anthropic-beta` values, as Python's `ANTHROPIC_BETA_HEADER_VALUES` lists them.
|
||||
pub mod beta {
|
||||
pub const CONTEXT_MANAGEMENT_2025_06_27: &str = "context-management-2025-06-27";
|
||||
pub const COMPACT_2026_01_12: &str = "compact-2026-01-12";
|
||||
pub const COMPACT_2026_09_04: &str = "compact-2026-09-04";
|
||||
pub const STRUCTURED_OUTPUT: &str = "structured-outputs-2025-11-13";
|
||||
pub const ADVANCED_TOOL_USE_2025_11_20: &str = "advanced-tool-use-2025-11-20";
|
||||
pub const FAST_MODE_2026_02_01: &str = "fast-mode-2026-02-01";
|
||||
pub const ADVISOR_TOOL_2026_03_01: &str = "advisor-tool-2026-03-01";
|
||||
pub const PER_TURN_CONTROL_2026_07_01: &str = "per-turn-control-2026-07-01";
|
||||
}
|
||||
|
||||
/// The `output_config.effort` levels, in ascending order.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum EffortLevel {
|
||||
Low,
|
||||
Medium,
|
||||
High,
|
||||
Xhigh,
|
||||
Max,
|
||||
}
|
||||
|
||||
impl EffortLevel {
|
||||
pub fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Self::Low => "low",
|
||||
Self::Medium => "medium",
|
||||
Self::High => "high",
|
||||
Self::Xhigh => "xhigh",
|
||||
Self::Max => "max",
|
||||
}
|
||||
}
|
||||
|
||||
pub fn parse(value: &str) -> Option<Self> {
|
||||
match value {
|
||||
"low" => Some(Self::Low),
|
||||
"medium" => Some(Self::Medium),
|
||||
"high" => Some(Self::High),
|
||||
"xhigh" => Some(Self::Xhigh),
|
||||
"max" => Some(Self::Max),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Which `reasoning_effort` tiers the cost map flags as `supports_<tier>_reasoning_effort`.
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct SupportedEffortTiers {
|
||||
#[serde(default)]
|
||||
pub minimal: bool,
|
||||
#[serde(default)]
|
||||
pub low: bool,
|
||||
#[serde(default)]
|
||||
pub medium: bool,
|
||||
#[serde(default)]
|
||||
pub high: bool,
|
||||
#[serde(default)]
|
||||
pub xhigh: bool,
|
||||
#[serde(default)]
|
||||
pub max: bool,
|
||||
}
|
||||
|
||||
impl SupportedEffortTiers {
|
||||
pub fn any(self) -> bool {
|
||||
self.minimal || self.low || self.medium || self.high || self.xhigh || self.max
|
||||
}
|
||||
}
|
||||
|
||||
/// The cost-map facts about a model that the Messages transformation branches on. The
|
||||
/// host resolves them the way Python's `AnthropicModelInfo._supports_model_capability`
|
||||
/// does, under the caller's provider and with the fallback generalizations applied.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct AnthropicModelCapabilities {
|
||||
#[serde(default)]
|
||||
pub supports_reasoning: bool,
|
||||
#[serde(default)]
|
||||
pub supports_adaptive_thinking: bool,
|
||||
#[serde(default)]
|
||||
pub thinking_always_on: bool,
|
||||
#[serde(default)]
|
||||
pub supports_legacy_thinking: bool,
|
||||
#[serde(default)]
|
||||
pub supports_output_config: bool,
|
||||
/// Whether `temperature`, `top_p` and `top_k` are accepted. Claude 4.7+ removed them.
|
||||
#[serde(default = "default_true")]
|
||||
pub supports_sampling_params: bool,
|
||||
/// Whether the model takes `speed` (fast mode), a direct Anthropic API feature.
|
||||
#[serde(default)]
|
||||
pub supports_speed: bool,
|
||||
#[serde(default)]
|
||||
pub effort_tiers: SupportedEffortTiers,
|
||||
}
|
||||
|
||||
fn default_true() -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
impl Default for AnthropicModelCapabilities {
|
||||
/// An unmapped model: no reasoning features, sampling params still accepted.
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
supports_reasoning: false,
|
||||
supports_adaptive_thinking: false,
|
||||
thinking_always_on: false,
|
||||
supports_legacy_thinking: false,
|
||||
supports_output_config: false,
|
||||
supports_sampling_params: true,
|
||||
supports_speed: false,
|
||||
effort_tiers: SupportedEffortTiers::default(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl AnthropicModelCapabilities {
|
||||
pub fn supports_effort_tier(&self, level: EffortLevel) -> bool {
|
||||
match level {
|
||||
EffortLevel::Low => self.effort_tiers.low,
|
||||
EffortLevel::Medium => self.effort_tiers.medium,
|
||||
EffortLevel::High => self.effort_tiers.high,
|
||||
EffortLevel::Xhigh => self.effort_tiers.xhigh,
|
||||
EffortLevel::Max => self.effort_tiers.max,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether the model accepts `output_config.effort` at all, as Python's
|
||||
/// `_model_supports_effort_param` decides it.
|
||||
pub fn supports_effort_param(&self) -> bool {
|
||||
self.supports_output_config || self.effort_tiers.any()
|
||||
}
|
||||
|
||||
/// Python's `_validate_effort_for_model`: `None` when the level is allowed, else the
|
||||
/// 400 message.
|
||||
pub fn effort_level_rejection(&self, effort: &str, model: &str) -> Option<String> {
|
||||
match effort {
|
||||
"max" if !(self.supports_adaptive_thinking || self.effort_tiers.max) => Some(format!(
|
||||
"effort='max' is not supported by this model. Got model: {model}"
|
||||
)),
|
||||
"xhigh" if !self.effort_tiers.xhigh => Some(format!(
|
||||
"effort='xhigh' is not supported by this model. Got model: {model}"
|
||||
)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_anthropic_oauth_key(value: &str) -> bool {
|
||||
value
|
||||
.strip_prefix("Bearer ")
|
||||
.unwrap_or(value)
|
||||
.starts_with(ANTHROPIC_OAUTH_TOKEN_PREFIX)
|
||||
}
|
||||
|
||||
/// Merge one more value into a comma separated `anthropic-beta` header, sorted and
|
||||
/// deduplicated like Python's `_merge_beta_headers`.
|
||||
pub fn merge_beta_headers(existing: Option<&str>, new_beta: &str) -> String {
|
||||
join_beta_values(split_beta_values(existing).chain(std::iter::once(new_beta.to_string())))
|
||||
}
|
||||
|
||||
pub fn split_beta_values(header: Option<&str>) -> impl Iterator<Item = String> + '_ {
|
||||
header
|
||||
.into_iter()
|
||||
.flat_map(|value| value.split(','))
|
||||
.map(str::trim)
|
||||
.filter(|piece| !piece.is_empty())
|
||||
.map(str::to_string)
|
||||
}
|
||||
|
||||
pub fn join_beta_values(values: impl IntoIterator<Item = String>) -> String {
|
||||
let mut values: Vec<String> = values.into_iter().collect();
|
||||
values.sort();
|
||||
values.dedup();
|
||||
values.join(",")
|
||||
}
|
||||
|
||||
pub fn is_tool_search_used(tools: Option<&[Value]>) -> bool {
|
||||
tools.into_iter().flatten().any(|tool| {
|
||||
tool.get("type")
|
||||
.and_then(Value::as_str)
|
||||
.is_some_and(|tool_type| ANTHROPIC_TOOL_SEARCH_TOOL_TYPES.contains(&tool_type))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn has_advisor_tool(tools: Option<&[Value]>) -> bool {
|
||||
tools
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.any(|tool| tool.get("type").and_then(Value::as_str) == Some(ANTHROPIC_ADVISOR_TOOL_TYPE))
|
||||
}
|
||||
|
||||
/// Whether the request already carries native compaction: a `compaction` param or a
|
||||
/// signed `compaction` block in history.
|
||||
pub fn requires_native_compaction_beta(
|
||||
compaction: Option<&Value>,
|
||||
messages: &[AnthropicMessage],
|
||||
) -> bool {
|
||||
compaction.is_some()
|
||||
|| messages
|
||||
.iter()
|
||||
.flat_map(AnthropicMessage::blocks)
|
||||
.any(|block| {
|
||||
block.is_type("compaction")
|
||||
&& block.signature.as_deref().is_some_and(|s| !s.is_empty())
|
||||
})
|
||||
}
|
||||
|
||||
fn is_blank(text: Option<&str>) -> bool {
|
||||
text.is_none_or(|text| text.trim().is_empty())
|
||||
}
|
||||
|
||||
fn is_empty_text_block(block: &ContentBlock) -> bool {
|
||||
block.is_type("text") && is_blank(block.text.as_deref())
|
||||
}
|
||||
|
||||
/// A `thinking` block with no thinking text. Anthropic rejects it regardless of any
|
||||
/// signature; `redacted_thinking` blocks are a different type and never match.
|
||||
pub fn is_empty_thinking_block(block: &ContentBlock) -> bool {
|
||||
block.is_type("thinking") && is_blank(block.thinking.as_deref())
|
||||
}
|
||||
|
||||
fn retain_blocks(
|
||||
messages: Vec<AnthropicMessage>,
|
||||
keep: impl Fn(&ContentBlock) -> bool,
|
||||
) -> Vec<AnthropicMessage> {
|
||||
messages
|
||||
.into_iter()
|
||||
.filter_map(|message| match message.content {
|
||||
MessageContent::Text(_) => Some(message),
|
||||
MessageContent::Blocks(ref blocks) => {
|
||||
let kept: Vec<ContentBlock> =
|
||||
blocks.iter().filter(|block| keep(block)).cloned().collect();
|
||||
if kept.len() == blocks.len() {
|
||||
return Some(message);
|
||||
}
|
||||
(!kept.is_empty()).then(|| message.with_blocks(kept))
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Drop empty `text` and `thinking` blocks. A message whose block list empties out is
|
||||
/// dropped with them, since Anthropic rejects an empty content array.
|
||||
pub fn strip_empty_content_blocks(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
|
||||
retain_blocks(messages, |block| {
|
||||
!is_empty_text_block(block) && !is_empty_thinking_block(block)
|
||||
})
|
||||
}
|
||||
|
||||
/// Rewrite a `tool_use` / `tool_result` id into Anthropic's `^[a-zA-Z0-9_-]+$` alphabet.
|
||||
pub fn normalize_anthropic_tool_use_id(raw_id: &str) -> String {
|
||||
let base = raw_id
|
||||
.split_once(THOUGHT_SIGNATURE_SEPARATOR)
|
||||
.map_or(raw_id, |(base, _)| base);
|
||||
let sanitized: String = base
|
||||
.chars()
|
||||
.map(|character| {
|
||||
if character.is_ascii_alphanumeric() || matches!(character, '_' | '-') {
|
||||
character
|
||||
} else {
|
||||
'_'
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
if sanitized.is_empty() {
|
||||
"tool_use_id".to_string()
|
||||
} else {
|
||||
sanitized
|
||||
}
|
||||
}
|
||||
|
||||
fn normalized_if_changed(raw_id: Option<&str>) -> Option<String> {
|
||||
let raw_id = raw_id?;
|
||||
let normalized = normalize_anthropic_tool_use_id(raw_id);
|
||||
(normalized != raw_id).then_some(normalized)
|
||||
}
|
||||
|
||||
fn sanitize_tool_use_id_block(block: ContentBlock) -> ContentBlock {
|
||||
match block.block_type.as_deref() {
|
||||
Some("tool_use" | "server_tool_use") => match normalized_if_changed(block.id.as_deref()) {
|
||||
Some(id) => ContentBlock {
|
||||
id: Some(id),
|
||||
..block
|
||||
},
|
||||
None => block,
|
||||
},
|
||||
Some("tool_result") => match normalized_if_changed(block.tool_use_id.as_deref()) {
|
||||
Some(tool_use_id) => ContentBlock {
|
||||
tool_use_id: Some(tool_use_id),
|
||||
..block
|
||||
},
|
||||
None => block,
|
||||
},
|
||||
_ => block,
|
||||
}
|
||||
}
|
||||
|
||||
fn map_blocks(
|
||||
messages: Vec<AnthropicMessage>,
|
||||
rewrite: impl Fn(Vec<ContentBlock>) -> Vec<ContentBlock>,
|
||||
) -> Vec<AnthropicMessage> {
|
||||
messages
|
||||
.into_iter()
|
||||
.map(|message| match message.content {
|
||||
MessageContent::Blocks(blocks) => AnthropicMessage {
|
||||
content: MessageContent::Blocks(rewrite(blocks)),
|
||||
..message
|
||||
},
|
||||
MessageContent::Text(_) => message,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Rewrite tool ids replayed from another provider (`functions.Bash:0`) into ids Anthropic
|
||||
/// accepts.
|
||||
pub fn sanitize_tool_use_ids(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
|
||||
map_blocks(messages, |blocks| {
|
||||
blocks.into_iter().map(sanitize_tool_use_id_block).collect()
|
||||
})
|
||||
}
|
||||
|
||||
/// Drop `provider_specific_fields` from every block; it is LiteLLM's own annotation.
|
||||
pub fn strip_provider_specific_fields(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
|
||||
map_blocks(messages, |blocks| {
|
||||
blocks
|
||||
.into_iter()
|
||||
.map(|block| ContentBlock {
|
||||
provider_specific_fields: None,
|
||||
..block
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
}
|
||||
|
||||
/// A thinking or redacted_thinking block carrying Responses API encrypted reasoning that
|
||||
/// only the bridge which minted it can read back.
|
||||
pub fn is_encrypted_reasoning_block(block: &ContentBlock) -> bool {
|
||||
let field = match block.block_type.as_deref() {
|
||||
Some("thinking") => block.signature.as_deref(),
|
||||
Some("redacted_thinking") => block.data.as_deref(),
|
||||
_ => None,
|
||||
};
|
||||
field.is_some_and(|value| value.starts_with(ENCRYPTED_REASONING_SIGNATURE_PREFIX))
|
||||
}
|
||||
|
||||
/// Drop the bridge-tagged reasoning blocks Anthropic cannot verify; its own signed
|
||||
/// blocks stay.
|
||||
pub fn strip_encrypted_reasoning_blocks(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
|
||||
retain_blocks(messages, |block| !is_encrypted_reasoning_block(block))
|
||||
}
|
||||
|
||||
fn is_advisor_use(block: &ContentBlock) -> bool {
|
||||
block.is_type("server_tool_use")
|
||||
&& block.name.as_deref() == Some("advisor")
|
||||
&& block.id.as_deref().is_some_and(|id| !id.is_empty())
|
||||
}
|
||||
|
||||
/// Drop `server_tool_use(name=advisor)` and `advisor_tool_result` blocks from assistant
|
||||
/// turns. Anthropic rejects advisor history when the advisor tool is not in `tools`.
|
||||
pub fn strip_advisor_blocks(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
|
||||
messages
|
||||
.into_iter()
|
||||
.map(|message| {
|
||||
if message.role != "assistant" {
|
||||
return message;
|
||||
}
|
||||
let MessageContent::Blocks(blocks) = &message.content else {
|
||||
return message;
|
||||
};
|
||||
let advisor_ids: Vec<&str> = blocks
|
||||
.iter()
|
||||
.filter(|block| is_advisor_use(block))
|
||||
.filter_map(|block| block.id.as_deref())
|
||||
.collect();
|
||||
if advisor_ids.is_empty() {
|
||||
return message;
|
||||
}
|
||||
let kept: Vec<ContentBlock> = blocks
|
||||
.iter()
|
||||
.filter(|block| {
|
||||
let is_result = block.is_type("advisor_tool_result")
|
||||
&& block
|
||||
.tool_use_id
|
||||
.as_deref()
|
||||
.is_some_and(|id| advisor_ids.contains(&id));
|
||||
!is_advisor_use(block) && !is_result
|
||||
})
|
||||
.cloned()
|
||||
.collect();
|
||||
message.with_blocks(kept)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct ReplayedWebSearchResult {
|
||||
#[serde(default)]
|
||||
url: String,
|
||||
#[serde(default)]
|
||||
title: String,
|
||||
#[serde(default)]
|
||||
snippet: String,
|
||||
#[serde(default)]
|
||||
encrypted_content: String,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
#[serde(tag = "type")]
|
||||
enum ReplayedWebSearchContent {
|
||||
#[serde(rename = "web_search_tool_result_error")]
|
||||
Error {
|
||||
#[serde(default)]
|
||||
error_code: String,
|
||||
},
|
||||
}
|
||||
|
||||
enum WebSearchResults {
|
||||
Results(Vec<ReplayedWebSearchResult>),
|
||||
Error(String),
|
||||
}
|
||||
|
||||
/// The results of a replayed `web_search_tool_result` block that carries no
|
||||
/// `encrypted_content`, else `None` for anything Anthropic itself issued.
|
||||
fn flattenable_web_search_results(block: &ContentBlock) -> Option<WebSearchResults> {
|
||||
if !block.is_type("web_search_tool_result") || block.tool_use_id.is_none() {
|
||||
return None;
|
||||
}
|
||||
match block.content.as_ref()? {
|
||||
Value::Array(items) => {
|
||||
let results = items
|
||||
.iter()
|
||||
.map(|item| {
|
||||
(item.get("type").and_then(Value::as_str) == Some("web_search_result"))
|
||||
.then(|| {
|
||||
serde_json::from_value::<ReplayedWebSearchResult>(item.clone()).ok()
|
||||
})
|
||||
.flatten()
|
||||
})
|
||||
.collect::<Option<Vec<_>>>()?;
|
||||
if results
|
||||
.iter()
|
||||
.any(|result| !result.encrypted_content.is_empty())
|
||||
{
|
||||
return None;
|
||||
}
|
||||
Some(WebSearchResults::Results(results))
|
||||
}
|
||||
error @ Value::Object(_) => match serde_json::from_value(error.clone()).ok()? {
|
||||
ReplayedWebSearchContent::Error { error_code } => {
|
||||
Some(WebSearchResults::Error(error_code))
|
||||
}
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn render_web_search_results(query: &str, results: &WebSearchResults) -> String {
|
||||
let header = if query.is_empty() {
|
||||
"Web search results:".to_string()
|
||||
} else {
|
||||
format!("Web search results for '{query}':")
|
||||
};
|
||||
match results {
|
||||
WebSearchResults::Error(code) => {
|
||||
let code = if code.is_empty() { "unavailable" } else { code };
|
||||
format!("{header}\n\nSearch failed: {code}")
|
||||
}
|
||||
WebSearchResults::Results(results) if results.is_empty() => {
|
||||
format!("{header}\n\nNo results were returned.")
|
||||
}
|
||||
WebSearchResults::Results(results) => {
|
||||
let body = results
|
||||
.iter()
|
||||
.map(|result| {
|
||||
[
|
||||
(!result.title.is_empty()).then(|| format!("Title: {}", result.title)),
|
||||
(!result.url.is_empty()).then(|| format!("URL: {}", result.url)),
|
||||
(!result.snippet.is_empty())
|
||||
.then(|| format!("Snippet: {}", result.snippet)),
|
||||
]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n")
|
||||
})
|
||||
.filter(|entry| !entry.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n\n");
|
||||
if body.is_empty() {
|
||||
header
|
||||
} else {
|
||||
format!("{header}\n\n{body}")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn server_tool_use_query(block: &ContentBlock) -> Option<(&str, &str)> {
|
||||
if !block.is_type("server_tool_use") {
|
||||
return None;
|
||||
}
|
||||
let id = block.id.as_deref()?;
|
||||
let query = block
|
||||
.input
|
||||
.as_ref()
|
||||
.and_then(|input| input.get("query"))
|
||||
.and_then(Value::as_str)
|
||||
.unwrap_or("");
|
||||
Some((id, query))
|
||||
}
|
||||
|
||||
fn flatten_web_search_results_in_blocks(blocks: Vec<ContentBlock>) -> Vec<ContentBlock> {
|
||||
let flattenable: Vec<(&str, WebSearchResults)> = blocks
|
||||
.iter()
|
||||
.filter_map(|block| {
|
||||
let results = flattenable_web_search_results(block)?;
|
||||
Some((block.tool_use_id.as_deref()?, results))
|
||||
})
|
||||
.collect();
|
||||
if flattenable.is_empty() {
|
||||
return blocks;
|
||||
}
|
||||
let queries: Vec<(&str, &str)> = blocks.iter().filter_map(server_tool_use_query).collect();
|
||||
let rewritten: Vec<ContentBlock> = blocks
|
||||
.iter()
|
||||
.filter_map(|block| {
|
||||
if let Some((tool_use_id, results)) = block
|
||||
.tool_use_id
|
||||
.as_deref()
|
||||
.and_then(|id| flattenable.iter().find(|(flat_id, _)| *flat_id == id))
|
||||
.filter(|_| block.is_type("web_search_tool_result"))
|
||||
{
|
||||
let query = queries
|
||||
.iter()
|
||||
.find(|(id, _)| id == tool_use_id)
|
||||
.map_or("", |(_, query)| query);
|
||||
return Some(ContentBlock::text(render_web_search_results(
|
||||
query, results,
|
||||
)));
|
||||
}
|
||||
if let Some((id, _)) = server_tool_use_query(block)
|
||||
&& flattenable.iter().any(|(flat_id, _)| *flat_id == id)
|
||||
{
|
||||
return None;
|
||||
}
|
||||
Some(block.clone())
|
||||
})
|
||||
.collect();
|
||||
rewritten
|
||||
}
|
||||
|
||||
/// Rewrite replayed `web_search_tool_result` blocks that carry no `encrypted_content`
|
||||
/// (LiteLLM synthesized them) into plain text so Anthropic does not reject the history.
|
||||
pub fn flatten_unencrypted_web_search_results(
|
||||
messages: Vec<AnthropicMessage>,
|
||||
) -> Vec<AnthropicMessage> {
|
||||
map_blocks(messages, flatten_web_search_results_in_blocks)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use rstest::rstest;
|
||||
use serde_json::json;
|
||||
|
||||
use super::*;
|
||||
|
||||
fn message(role: &str, content: Value) -> AnthropicMessage {
|
||||
serde_json::from_value(json!({"role": role, "content": content})).unwrap()
|
||||
}
|
||||
|
||||
fn wire(messages: &[AnthropicMessage]) -> Value {
|
||||
serde_json::to_value(messages).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_empty_content_blocks_drops_blank_text_and_thinking_and_empty_messages() {
|
||||
let messages = vec![
|
||||
message(
|
||||
"assistant",
|
||||
json!([
|
||||
{"type": "text", "text": " "},
|
||||
{"type": "thinking", "thinking": "", "signature": "sig"},
|
||||
{"type": "redacted_thinking", "data": "opaque"},
|
||||
{"type": "tool_use", "id": "t1", "name": "f", "input": {}}
|
||||
]),
|
||||
),
|
||||
message("user", json!([{"type": "text", "text": ""}])),
|
||||
message("user", json!("plain")),
|
||||
];
|
||||
assert_eq!(
|
||||
wire(&strip_empty_content_blocks(messages)),
|
||||
json!([
|
||||
{"role": "assistant", "content": [
|
||||
{"type": "redacted_thinking", "data": "opaque"},
|
||||
{"type": "tool_use", "id": "t1", "name": "f", "input": {}}
|
||||
]},
|
||||
{"role": "user", "content": "plain"}
|
||||
])
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case("functions.Bash:0", "functions_Bash_0")]
|
||||
#[case("call_1__thought__abc", "call_1")]
|
||||
#[case("toolu_01", "toolu_01")]
|
||||
#[case("::", "__")]
|
||||
#[case("", "tool_use_id")]
|
||||
fn normalize_tool_use_id_matches_python(#[case] raw: &str, #[case] expected: &str) {
|
||||
assert_eq!(normalize_anthropic_tool_use_id(raw), expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sanitize_tool_use_ids_rewrites_use_and_result_ids_only() {
|
||||
let messages = vec![
|
||||
message(
|
||||
"assistant",
|
||||
json!([{"type": "tool_use", "id": "functions.Bash:0", "name": "Bash", "input": {}}]),
|
||||
),
|
||||
message(
|
||||
"user",
|
||||
json!([{"type": "tool_result", "tool_use_id": "functions.Bash:0", "content": "ok"}]),
|
||||
),
|
||||
message(
|
||||
"user",
|
||||
json!([{"type": "text", "text": "id: functions.Bash:0"}]),
|
||||
),
|
||||
];
|
||||
let sanitized = wire(&sanitize_tool_use_ids(messages));
|
||||
assert_eq!(sanitized[0]["content"][0]["id"], "functions_Bash_0");
|
||||
assert_eq!(
|
||||
sanitized[1]["content"][0]["tool_use_id"],
|
||||
"functions_Bash_0"
|
||||
);
|
||||
assert_eq!(sanitized[2]["content"][0]["text"], "id: functions.Bash:0");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_provider_specific_fields_leaves_other_keys() {
|
||||
let messages = vec![message(
|
||||
"assistant",
|
||||
json!([{"type": "thinking", "thinking": "hm", "signature": "s", "provider_specific_fields": {"a": 1}}]),
|
||||
)];
|
||||
assert_eq!(
|
||||
wire(&strip_provider_specific_fields(messages)),
|
||||
json!([{"role": "assistant", "content": [{"type": "thinking", "thinking": "hm", "signature": "s"}]}])
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_encrypted_reasoning_blocks_keeps_anthropic_signed_blocks() {
|
||||
let messages = vec![
|
||||
message(
|
||||
"assistant",
|
||||
json!([
|
||||
{"type": "thinking", "thinking": "a", "signature": "litellm_encrypted_reasoning:xyz"},
|
||||
{"type": "redacted_thinking", "data": "litellm_encrypted_reasoning:xyz"},
|
||||
{"type": "thinking", "thinking": "b", "signature": "anthropic-sig"},
|
||||
]),
|
||||
),
|
||||
message(
|
||||
"assistant",
|
||||
json!([{"type": "thinking", "thinking": "a", "signature": "litellm_encrypted_reasoning:xyz"}]),
|
||||
),
|
||||
];
|
||||
assert_eq!(
|
||||
wire(&strip_encrypted_reasoning_blocks(messages)),
|
||||
json!([{"role": "assistant", "content": [{"type": "thinking", "thinking": "b", "signature": "anthropic-sig"}]}])
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_advisor_blocks_removes_matched_pairs_from_assistant_turns_only() {
|
||||
let messages = vec![
|
||||
message(
|
||||
"assistant",
|
||||
json!([
|
||||
{"type": "server_tool_use", "id": "adv_1", "name": "advisor", "input": {}},
|
||||
{"type": "advisor_tool_result", "tool_use_id": "adv_1", "content": "advice"},
|
||||
{"type": "advisor_tool_result", "tool_use_id": "other", "content": "kept"},
|
||||
{"type": "text", "text": "answer"}
|
||||
]),
|
||||
),
|
||||
message(
|
||||
"user",
|
||||
json!([{"type": "server_tool_use", "id": "adv_2", "name": "advisor", "input": {}}]),
|
||||
),
|
||||
];
|
||||
let stripped = wire(&strip_advisor_blocks(messages));
|
||||
assert_eq!(
|
||||
stripped[0]["content"],
|
||||
json!([
|
||||
{"type": "advisor_tool_result", "tool_use_id": "other", "content": "kept"},
|
||||
{"type": "text", "text": "answer"}
|
||||
])
|
||||
);
|
||||
assert_eq!(stripped[1]["content"].as_array().unwrap().len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn flatten_web_search_results_rewrites_unencrypted_results_and_drops_their_server_tool_use() {
|
||||
let messages = vec![message(
|
||||
"assistant",
|
||||
json!([
|
||||
{"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "rust"}},
|
||||
{"type": "web_search_tool_result", "tool_use_id": "s1", "content": [
|
||||
{"type": "web_search_result", "url": "https://r", "title": "Rust", "snippet": "fast"},
|
||||
{"type": "web_search_result", "url": "https://q", "title": "", "snippet": ""}
|
||||
]},
|
||||
{"type": "server_tool_use", "id": "s2", "name": "web_search", "input": {"query": "real"}},
|
||||
{"type": "web_search_tool_result", "tool_use_id": "s2", "content": [
|
||||
{"type": "web_search_result", "url": "https://a", "title": "A", "snippet": "b", "encrypted_content": "enc"}
|
||||
]},
|
||||
{"type": "text", "text": "done"}
|
||||
]),
|
||||
)];
|
||||
assert_eq!(
|
||||
wire(&flatten_unencrypted_web_search_results(messages))[0]["content"],
|
||||
json!([
|
||||
{"type": "text", "text": "Web search results for 'rust':\n\nTitle: Rust\nURL: https://r\nSnippet: fast\n\nURL: https://q"},
|
||||
{"type": "server_tool_use", "id": "s2", "name": "web_search", "input": {"query": "real"}},
|
||||
{"type": "web_search_tool_result", "tool_use_id": "s2", "content": [
|
||||
{"type": "web_search_result", "url": "https://a", "title": "A", "snippet": "b", "encrypted_content": "enc"}
|
||||
]},
|
||||
{"type": "text", "text": "done"}
|
||||
])
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn flatten_web_search_results_renders_errors_and_empty_results() {
|
||||
let messages = vec![message(
|
||||
"assistant",
|
||||
json!([
|
||||
{"type": "web_search_tool_result", "tool_use_id": "e1", "content": {"type": "web_search_tool_result_error", "error_code": "max_uses"}},
|
||||
{"type": "web_search_tool_result", "tool_use_id": "e2", "content": []}
|
||||
]),
|
||||
)];
|
||||
assert_eq!(
|
||||
wire(&flatten_unencrypted_web_search_results(messages))[0]["content"],
|
||||
json!([
|
||||
{"type": "text", "text": "Web search results:\n\nSearch failed: max_uses"},
|
||||
{"type": "text", "text": "Web search results:\n\nNo results were returned."}
|
||||
])
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn beta_header_merging_sorts_and_dedupes() {
|
||||
assert_eq!(merge_beta_headers(None, "b"), "b");
|
||||
assert_eq!(merge_beta_headers(Some("b, a ,b"), "c"), "a,b,c");
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case("sk-ant-oat01-x", true)]
|
||||
#[case("Bearer sk-ant-oat01-x", true)]
|
||||
#[case("sk-ant-api03-x", false)]
|
||||
fn oauth_key_detection(#[case] value: &str, #[case] expected: bool) {
|
||||
assert_eq!(is_anthropic_oauth_key(value), expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compaction_beta_requires_param_or_signed_block() {
|
||||
let signed = vec![message(
|
||||
"assistant",
|
||||
json!([{"type": "compaction", "content": "c", "signature": "s"}]),
|
||||
)];
|
||||
let unsigned = vec![message(
|
||||
"assistant",
|
||||
json!([{"type": "compaction", "content": "c"}]),
|
||||
)];
|
||||
assert!(requires_native_compaction_beta(None, &signed));
|
||||
assert!(!requires_native_compaction_beta(None, &unsigned));
|
||||
assert!(requires_native_compaction_beta(Some(&json!({})), &unsigned));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn capability_effort_gates_match_python() {
|
||||
let opus_4_5 = AnthropicModelCapabilities {
|
||||
supports_reasoning: true,
|
||||
supports_output_config: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(opus_4_5.supports_effort_param());
|
||||
assert!(opus_4_5.effort_level_rejection("high", "m").is_none());
|
||||
assert!(opus_4_5.effort_level_rejection("xhigh", "m").is_some());
|
||||
assert!(opus_4_5.effort_level_rejection("max", "m").is_some());
|
||||
let adaptive = AnthropicModelCapabilities {
|
||||
supports_adaptive_thinking: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(adaptive.effort_level_rejection("max", "m").is_none());
|
||||
assert!(adaptive.effort_level_rejection("xhigh", "m").is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn capabilities_deserialize_with_python_defaults() {
|
||||
let parsed: AnthropicModelCapabilities = serde_json::from_value(json!({})).unwrap();
|
||||
assert!(parsed.supports_sampling_params);
|
||||
assert!(!parsed.supports_reasoning);
|
||||
let flagged: AnthropicModelCapabilities = serde_json::from_value(
|
||||
json!({"effort_tiers": {"xhigh": true}, "supports_sampling_params": false}),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(flagged.effort_tiers.xhigh);
|
||||
assert!(!flagged.supports_sampling_params);
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,332 @@
|
|||
//! Request headers for the direct Anthropic Messages API: credential resolution, including
|
||||
//! OAuth tokens, and the `anthropic-beta` values a request's features call for.
|
||||
|
||||
use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest;
|
||||
use serde_json::Value;
|
||||
|
||||
use crate::anthropic::{
|
||||
ANTHROPIC_OAUTH_TOKEN_PREFIX,
|
||||
common_utils::{
|
||||
ANTHROPIC_OAUTH_BETA_HEADER, beta, has_advisor_tool, is_anthropic_oauth_key,
|
||||
is_tool_search_used, join_beta_values, requires_native_compaction_beta, split_beta_values,
|
||||
},
|
||||
};
|
||||
|
||||
const ANTHROPIC_API_KEY_ENV: &str = "ANTHROPIC_API_KEY";
|
||||
const ANTHROPIC_AUTH_TOKEN_ENV: &str = "ANTHROPIC_AUTH_TOKEN";
|
||||
const BETA_HEADER: &str = "anthropic-beta";
|
||||
const AUTHORIZATION: &str = "authorization";
|
||||
const API_KEY_HEADER: &str = "x-api-key";
|
||||
const DIRECT_BROWSER_ACCESS_HEADER: &str = "anthropic-dangerous-direct-browser-access";
|
||||
|
||||
pub type Headers = Vec<(String, String)>;
|
||||
|
||||
fn header_value<'a>(headers: &'a [(String, String)], name: &str) -> Option<&'a str> {
|
||||
headers
|
||||
.iter()
|
||||
.find(|(header, _)| header.eq_ignore_ascii_case(name))
|
||||
.map(|(_, value)| value.as_str())
|
||||
}
|
||||
|
||||
fn without(headers: Headers, names: &[&str]) -> Headers {
|
||||
headers
|
||||
.into_iter()
|
||||
.filter(|(header, _)| !names.iter().any(|name| header.eq_ignore_ascii_case(name)))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn with_oauth_bearer(headers: Headers, bearer: String) -> Headers {
|
||||
let beta = merge_oauth_beta(header_value(&headers, BETA_HEADER));
|
||||
without(headers, &[API_KEY_HEADER, AUTHORIZATION, BETA_HEADER])
|
||||
.into_iter()
|
||||
.chain([
|
||||
(AUTHORIZATION.to_string(), bearer),
|
||||
(BETA_HEADER.to_string(), beta),
|
||||
(DIRECT_BROWSER_ACCESS_HEADER.to_string(), "true".to_string()),
|
||||
])
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn merge_oauth_beta(existing: Option<&str>) -> String {
|
||||
join_beta_values(
|
||||
split_beta_values(existing).chain(std::iter::once(ANTHROPIC_OAUTH_BETA_HEADER.to_string())),
|
||||
)
|
||||
}
|
||||
|
||||
fn non_empty(value: Option<&str>) -> Option<&str> {
|
||||
value.map(str::trim).filter(|value| !value.is_empty())
|
||||
}
|
||||
|
||||
/// Python's `validate_anthropic_messages_environment` credential steps: an OAuth token in
|
||||
/// the forwarded `authorization` header or in `api_key` is the whole credential; otherwise a
|
||||
/// forwarded auth header is kept, else the key (`ANTHROPIC_API_KEY`) or the bearer token
|
||||
/// (`ANTHROPIC_AUTH_TOKEN`) is resolved.
|
||||
pub fn authenticate(
|
||||
headers: Headers,
|
||||
api_key: Option<&str>,
|
||||
env_lookup: &dyn Fn(&str) -> Option<String>,
|
||||
) -> Result<Headers, litellm_auth::Error> {
|
||||
if let Some(forwarded) = header_value(&headers, AUTHORIZATION)
|
||||
&& forwarded
|
||||
.strip_prefix("Bearer ")
|
||||
.is_some_and(|token| token.starts_with(ANTHROPIC_OAUTH_TOKEN_PREFIX))
|
||||
{
|
||||
let bearer = forwarded.to_string();
|
||||
return Ok(with_oauth_bearer(headers, bearer));
|
||||
}
|
||||
if let Some(key) = api_key.filter(|key| key.starts_with(ANTHROPIC_OAUTH_TOKEN_PREFIX)) {
|
||||
return Ok(with_oauth_bearer(headers, format!("Bearer {key}")));
|
||||
}
|
||||
if header_value(&headers, API_KEY_HEADER).is_some()
|
||||
|| header_value(&headers, AUTHORIZATION).is_some()
|
||||
{
|
||||
return Ok(headers);
|
||||
}
|
||||
let resolved_key = non_empty(api_key)
|
||||
.map(str::to_string)
|
||||
.or_else(|| env_lookup(ANTHROPIC_API_KEY_ENV).filter(|value| !value.trim().is_empty()));
|
||||
let auth = match resolved_key {
|
||||
Some(key) if is_anthropic_oauth_key(&key) => {
|
||||
(AUTHORIZATION.to_string(), format!("Bearer {key}"))
|
||||
}
|
||||
Some(key) => (API_KEY_HEADER.to_string(), key),
|
||||
None => match env_lookup(ANTHROPIC_AUTH_TOKEN_ENV).filter(|value| !value.trim().is_empty())
|
||||
{
|
||||
Some(token) => (AUTHORIZATION.to_string(), format!("Bearer {token}")),
|
||||
None => {
|
||||
return Err(litellm_auth::Error::MissingApiKey {
|
||||
provider: "Anthropic",
|
||||
environment_variable: ANTHROPIC_API_KEY_ENV,
|
||||
});
|
||||
}
|
||||
},
|
||||
};
|
||||
Ok(headers.into_iter().chain([auth]).collect())
|
||||
}
|
||||
|
||||
fn context_management_betas(
|
||||
context_management: Option<&Value>,
|
||||
) -> impl Iterator<Item = &'static str> {
|
||||
let edits = context_management
|
||||
.and_then(|value| value.get("edits"))
|
||||
.and_then(Value::as_array)
|
||||
.map(Vec::as_slice)
|
||||
.unwrap_or(&[]);
|
||||
let (compact, other) = edits.iter().fold((false, false), |(compact, other), edit| {
|
||||
match edit.get("type").and_then(Value::as_str) {
|
||||
Some("compact_20260112") => (true, other),
|
||||
_ => (compact, true),
|
||||
}
|
||||
});
|
||||
compact
|
||||
.then_some(beta::COMPACT_2026_01_12)
|
||||
.into_iter()
|
||||
.chain(other.then_some(beta::CONTEXT_MANAGEMENT_2025_06_27))
|
||||
}
|
||||
|
||||
fn uses_structured_output(request: &AnthropicMessagesRequest) -> bool {
|
||||
request.output_format.is_some()
|
||||
|| request
|
||||
.output_config
|
||||
.as_ref()
|
||||
.and_then(|config| config.get("format"))
|
||||
.is_some_and(|format| !format.is_null())
|
||||
}
|
||||
|
||||
fn messages_carry_output_config(request: &AnthropicMessagesRequest) -> bool {
|
||||
request
|
||||
.messages
|
||||
.iter()
|
||||
.any(|message| message.extra.contains_key("output_config"))
|
||||
}
|
||||
|
||||
/// The `anthropic-beta` values the request's features need, as Python's
|
||||
/// `_update_headers_with_anthropic_beta` derives them for the direct API.
|
||||
pub fn feature_betas(request: &AnthropicMessagesRequest) -> Vec<&'static str> {
|
||||
let tools = request.tools.as_deref();
|
||||
[
|
||||
requires_native_compaction_beta(request.compaction.as_ref(), &request.messages)
|
||||
.then_some(beta::COMPACT_2026_09_04),
|
||||
uses_structured_output(request).then_some(beta::STRUCTURED_OUTPUT),
|
||||
(request.speed.as_deref() == Some("fast")).then_some(beta::FAST_MODE_2026_02_01),
|
||||
messages_carry_output_config(request).then_some(beta::PER_TURN_CONTROL_2026_07_01),
|
||||
has_advisor_tool(tools).then_some(beta::ADVISOR_TOOL_2026_03_01),
|
||||
is_tool_search_used(tools).then_some(beta::ADVANCED_TOOL_USE_2025_11_20),
|
||||
]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.chain(context_management_betas(
|
||||
request.context_management.as_ref(),
|
||||
))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Merge the request's feature betas into the outgoing headers. Headers without any beta
|
||||
/// value are returned untouched.
|
||||
pub fn with_feature_betas(headers: Headers, request: &AnthropicMessagesRequest) -> Headers {
|
||||
let existing = split_beta_values(header_value(&headers, BETA_HEADER)).collect::<Vec<_>>();
|
||||
let features = feature_betas(request);
|
||||
if existing.is_empty() && features.is_empty() {
|
||||
return headers;
|
||||
}
|
||||
let merged = join_beta_values(
|
||||
existing
|
||||
.into_iter()
|
||||
.chain(features.into_iter().map(str::to_string)),
|
||||
);
|
||||
without(headers, &[BETA_HEADER])
|
||||
.into_iter()
|
||||
.chain([(BETA_HEADER.to_string(), merged)])
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use rstest::rstest;
|
||||
use serde_json::json;
|
||||
|
||||
use super::*;
|
||||
|
||||
fn request(fields: Value) -> AnthropicMessagesRequest {
|
||||
let mut body =
|
||||
json!({"model": "claude", "messages": [{"role": "user", "content": "Hello"}]});
|
||||
body.as_object_mut()
|
||||
.unwrap()
|
||||
.extend(fields.as_object().unwrap().clone());
|
||||
serde_json::from_value(body).unwrap()
|
||||
}
|
||||
|
||||
fn header(name: &str, value: &str) -> (String, String) {
|
||||
(name.to_string(), value.to_string())
|
||||
}
|
||||
|
||||
fn no_env(_: &str) -> Option<String> {
|
||||
None
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn forwarded_oauth_bearer_replaces_the_key_and_adds_oauth_headers() {
|
||||
let headers = authenticate(
|
||||
vec![
|
||||
header("X-Api-Key", "sk-ant-api03-deployment"),
|
||||
header("Authorization", "Bearer sk-ant-oat01-token"),
|
||||
header("anthropic-beta", "web-search-2025-03-05"),
|
||||
],
|
||||
Some("sk-ant-api03-deployment"),
|
||||
&no_env,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(header_value(&headers, "x-api-key"), None);
|
||||
assert_eq!(
|
||||
header_value(&headers, "authorization"),
|
||||
Some("Bearer sk-ant-oat01-token")
|
||||
);
|
||||
assert_eq!(
|
||||
header_value(&headers, "anthropic-beta"),
|
||||
Some("oauth-2025-04-20,web-search-2025-03-05")
|
||||
);
|
||||
assert_eq!(
|
||||
header_value(&headers, DIRECT_BROWSER_ACCESS_HEADER),
|
||||
Some("true")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn oauth_api_key_authenticates_as_a_bearer() {
|
||||
let headers = authenticate(vec![], Some("sk-ant-oat01-token"), &no_env).unwrap();
|
||||
assert_eq!(
|
||||
header_value(&headers, "authorization"),
|
||||
Some("Bearer sk-ant-oat01-token")
|
||||
);
|
||||
assert_eq!(
|
||||
header_value(&headers, "anthropic-beta"),
|
||||
Some("oauth-2025-04-20")
|
||||
);
|
||||
assert_eq!(header_value(&headers, "x-api-key"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn forwarded_auth_headers_are_kept_without_resolving_a_key() {
|
||||
let forwarded = vec![header("Authorization", "Bearer some-proxy-token")];
|
||||
let headers = authenticate(forwarded.clone(), None, &no_env).unwrap();
|
||||
assert_eq!(headers, forwarded);
|
||||
let keyed = vec![header("x-api-key", "caller-key")];
|
||||
assert_eq!(
|
||||
authenticate(keyed.clone(), Some("sk-other"), &no_env).unwrap(),
|
||||
keyed
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case(Some("sk-param"), &[], "x-api-key", "sk-param")]
|
||||
#[case(Some(" "), &[("ANTHROPIC_API_KEY", "sk-env")], "x-api-key", "sk-env")]
|
||||
#[case(None, &[("ANTHROPIC_AUTH_TOKEN", "tok")], "authorization", "Bearer tok")]
|
||||
#[case(None, &[("ANTHROPIC_API_KEY", "sk-ant-oat01-env")], "authorization", "Bearer sk-ant-oat01-env")]
|
||||
fn credential_resolution_order_matches_python(
|
||||
#[case] api_key: Option<&str>,
|
||||
#[case] env: &[(&str, &str)],
|
||||
#[case] expected_header: &str,
|
||||
#[case] expected_value: &str,
|
||||
) {
|
||||
let lookup = |name: &str| {
|
||||
env.iter()
|
||||
.find(|(key, _)| *key == name)
|
||||
.map(|(_, value)| value.to_string())
|
||||
};
|
||||
let headers = authenticate(vec![], api_key, &lookup).unwrap();
|
||||
assert_eq!(headers, vec![header(expected_header, expected_value)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_credentials_are_an_auth_error() {
|
||||
assert!(matches!(
|
||||
authenticate(vec![], None, &no_env),
|
||||
Err(litellm_auth::Error::MissingApiKey { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case(json!({}), &[])]
|
||||
#[case(json!({"output_format": {"type": "json_schema"}}), &[beta::STRUCTURED_OUTPUT])]
|
||||
#[case(json!({"output_config": {"format": {"type": "json_schema"}}}), &[beta::STRUCTURED_OUTPUT])]
|
||||
#[case(json!({"output_config": {"effort": "high"}}), &[])]
|
||||
#[case(json!({"speed": "fast"}), &[beta::FAST_MODE_2026_02_01])]
|
||||
#[case(json!({"speed": "standard"}), &[])]
|
||||
#[case(json!({"compaction": {"enabled": true}}), &[beta::COMPACT_2026_09_04])]
|
||||
#[case(json!({"tools": [{"type": "advisor_20260301"}]}), &[beta::ADVISOR_TOOL_2026_03_01])]
|
||||
#[case(json!({"tools": [{"type": "tool_search_tool_regex_20251119"}]}), &[beta::ADVANCED_TOOL_USE_2025_11_20])]
|
||||
#[case(
|
||||
json!({"context_management": {"edits": [{"type": "compact_20260112"}, {"type": "clear_tool_uses_20250919"}]}}),
|
||||
&[beta::COMPACT_2026_01_12, beta::CONTEXT_MANAGEMENT_2025_06_27]
|
||||
)]
|
||||
#[case(
|
||||
json!({"messages": [{"role": "user", "content": "hi", "output_config": {"effort": "low"}}]}),
|
||||
&[beta::PER_TURN_CONTROL_2026_07_01]
|
||||
)]
|
||||
fn feature_betas_follow_the_request(#[case] fields: Value, #[case] expected: &[&str]) {
|
||||
assert_eq!(feature_betas(&request(fields)), expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn feature_betas_merge_into_the_existing_header_sorted() {
|
||||
let headers = with_feature_betas(
|
||||
vec![header(
|
||||
"Anthropic-Beta",
|
||||
"web-search-2025-03-05, compact-2026-09-04",
|
||||
)],
|
||||
&request(json!({"speed": "fast"})),
|
||||
);
|
||||
assert_eq!(
|
||||
headers,
|
||||
vec![header(
|
||||
"anthropic-beta",
|
||||
"compact-2026-09-04,fast-mode-2026-02-01,web-search-2025-03-05"
|
||||
)]
|
||||
);
|
||||
let untouched = vec![header("x-api-key", "k")];
|
||||
assert_eq!(
|
||||
with_feature_betas(untouched.clone(), &request(json!({}))),
|
||||
untouched
|
||||
);
|
||||
}
|
||||
}
|
||||
|
|
@ -1,2 +1,4 @@
|
|||
pub mod headers;
|
||||
pub mod streaming_iterator;
|
||||
pub mod thinking;
|
||||
pub mod transformation;
|
||||
|
|
|
|||
|
|
@ -0,0 +1,788 @@
|
|||
//! Reasoning parameter translation for the Messages API: OpenAI-style `reasoning_effort`,
|
||||
//! the 4.6+ adaptive interface and the legacy `budget_tokens` interface, reshaped to what
|
||||
//! the target model accepts. Mirrors the thinking steps of Python's
|
||||
//! `AnthropicMessagesConfig.transform_anthropic_messages_request`.
|
||||
|
||||
use litellm_core_utils::settings::Lookup;
|
||||
use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest;
|
||||
use serde_json::{Map, Value, json};
|
||||
|
||||
use crate::{
|
||||
anthropic::common_utils::AnthropicModelCapabilities, base_llm::chat::transformation::Error,
|
||||
};
|
||||
|
||||
pub const ANTHROPIC_MIN_THINKING_BUDGET_TOKENS: u64 = 1024;
|
||||
|
||||
const EFFORT_NAMES: &str = "'minimal', 'low', 'medium', 'high', 'xhigh', 'max', 'none'";
|
||||
|
||||
/// The `budget_tokens` each `reasoning_effort` tier maps to. Every value honors the same
|
||||
/// `DEFAULT_REASONING_EFFORT_<TIER>_THINKING_BUDGET` environment override Python reads.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub struct ThinkingBudgets {
|
||||
pub minimal: u64,
|
||||
pub low: u64,
|
||||
pub medium: u64,
|
||||
pub high: u64,
|
||||
pub xhigh: u64,
|
||||
pub max: u64,
|
||||
}
|
||||
|
||||
impl Default for ThinkingBudgets {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
minimal: 128,
|
||||
low: 1024,
|
||||
medium: 2048,
|
||||
high: 4096,
|
||||
xhigh: 8192,
|
||||
max: 16384,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl ThinkingBudgets {
|
||||
pub fn from_lookup(env: &impl Lookup) -> Self {
|
||||
let defaults = Self::default();
|
||||
let tier = |name: &str, default: u64| {
|
||||
env.parsed::<u64>(&format!("DEFAULT_REASONING_EFFORT_{name}_THINKING_BUDGET"))
|
||||
.unwrap_or(default)
|
||||
};
|
||||
Self {
|
||||
minimal: tier("MINIMAL", defaults.minimal),
|
||||
low: tier("LOW", defaults.low),
|
||||
medium: tier("MEDIUM", defaults.medium),
|
||||
high: tier("HIGH", defaults.high),
|
||||
xhigh: tier("XHIGH", defaults.xhigh),
|
||||
max: tier("MAX", defaults.max),
|
||||
}
|
||||
}
|
||||
|
||||
fn for_effort(&self, reasoning_effort: &str) -> Option<u64> {
|
||||
match reasoning_effort {
|
||||
"low" => Some(self.low),
|
||||
"medium" => Some(self.medium),
|
||||
"high" => Some(self.high),
|
||||
"xhigh" => Some(self.xhigh),
|
||||
"max" => Some(self.max),
|
||||
"minimal" => Some(self.minimal.max(ANTHROPIC_MIN_THINKING_BUDGET_TOKENS)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Python's `_legacy_budget_to_effort`: the effort tier a legacy budget stands for.
|
||||
fn effort_for_budget(
|
||||
&self,
|
||||
budget_tokens: u64,
|
||||
capabilities: &AnthropicModelCapabilities,
|
||||
) -> &'static str {
|
||||
if budget_tokens >= self.xhigh && capabilities.effort_tiers.xhigh {
|
||||
return "xhigh";
|
||||
}
|
||||
if budget_tokens >= self.high {
|
||||
return "high";
|
||||
}
|
||||
if budget_tokens >= self.medium {
|
||||
return "medium";
|
||||
}
|
||||
"low"
|
||||
}
|
||||
}
|
||||
|
||||
/// Everything the thinking translation needs to know besides the request itself.
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||
pub struct ThinkingContext {
|
||||
pub capabilities: AnthropicModelCapabilities,
|
||||
pub budgets: ThinkingBudgets,
|
||||
}
|
||||
|
||||
fn bad_request(message: String) -> Error {
|
||||
Error::InvalidRequest(message)
|
||||
}
|
||||
|
||||
fn thinking_type(thinking: Option<&Value>) -> Option<&str> {
|
||||
thinking?.get("type")?.as_str()
|
||||
}
|
||||
|
||||
fn output_config_effort(output_config: Option<&Value>) -> Option<&str> {
|
||||
output_config?.get("effort")?.as_str()
|
||||
}
|
||||
|
||||
fn enabled_thinking(budget_tokens: u64) -> Value {
|
||||
json!({"type": "enabled", "budget_tokens": budget_tokens})
|
||||
}
|
||||
|
||||
/// Python's `AnthropicConfig._map_reasoning_effort`: `Ok(None)` for `none`, the adaptive
|
||||
/// shape for adaptive models, else a legacy budget.
|
||||
fn map_reasoning_effort(
|
||||
reasoning_effort: &str,
|
||||
context: &ThinkingContext,
|
||||
) -> Result<Option<Value>, Error> {
|
||||
if reasoning_effort == "none" {
|
||||
return Ok(None);
|
||||
}
|
||||
if context.capabilities.supports_adaptive_thinking {
|
||||
return Ok(Some(json!({"type": "adaptive", "display": "summarized"})));
|
||||
}
|
||||
context
|
||||
.budgets
|
||||
.for_effort(reasoning_effort)
|
||||
.map(|budget| Some(enabled_thinking(budget)))
|
||||
.ok_or_else(|| {
|
||||
bad_request(format!(
|
||||
"Unmapped reasoning effort: {reasoning_effort:?}. Must be one of: {EFFORT_NAMES}."
|
||||
))
|
||||
})
|
||||
}
|
||||
|
||||
/// Cap a legacy `budget_tokens` below `max_tokens`; `None` when even the minimum budget
|
||||
/// cannot fit and thinking should be dropped.
|
||||
fn cap_thinking_budget_to_max_tokens(thinking: Value, max_tokens: Option<u64>) -> Option<Value> {
|
||||
let (Some(max_tokens), Some(budget)) = (
|
||||
max_tokens,
|
||||
thinking.get("budget_tokens").and_then(Value::as_u64),
|
||||
) else {
|
||||
return Some(thinking);
|
||||
};
|
||||
if max_tokens <= ANTHROPIC_MIN_THINKING_BUDGET_TOKENS {
|
||||
return None;
|
||||
}
|
||||
if budget < max_tokens {
|
||||
return Some(thinking);
|
||||
}
|
||||
let thinking_type = thinking_type(Some(&thinking)).unwrap_or("enabled");
|
||||
Some(json!({"type": thinking_type, "budget_tokens": max_tokens - 1}))
|
||||
}
|
||||
|
||||
fn reasoning_effort_to_output_config_effort(reasoning_effort: &str) -> Option<&'static str> {
|
||||
match reasoning_effort {
|
||||
"low" | "minimal" => Some("low"),
|
||||
"medium" => Some("medium"),
|
||||
"high" => Some("high"),
|
||||
"xhigh" => Some("xhigh"),
|
||||
"max" => Some("max"),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn with_effort(output_config: Option<Value>, effort: &str, caller_wins: bool) -> Value {
|
||||
let mut config = match output_config {
|
||||
Some(Value::Object(config)) => config,
|
||||
_ => Map::new(),
|
||||
};
|
||||
if !caller_wins || !config.contains_key("effort") {
|
||||
config.insert("effort".to_string(), Value::String(effort.to_string()));
|
||||
}
|
||||
Value::Object(config)
|
||||
}
|
||||
|
||||
/// `reasoning_effort` becomes native `thinking` (and `output_config.effort` on adaptive
|
||||
/// models). Caller-supplied `thinking` / `output_config` win; `none` clears both.
|
||||
fn translate_reasoning_effort(
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &ThinkingContext,
|
||||
) -> Result<AnthropicMessagesRequest, Error> {
|
||||
let Some(reasoning_effort) = request.reasoning_effort.clone() else {
|
||||
return Ok(request);
|
||||
};
|
||||
let request = AnthropicMessagesRequest {
|
||||
reasoning_effort: None,
|
||||
..request
|
||||
};
|
||||
let Some(mapped) = map_reasoning_effort(&reasoning_effort, context)? else {
|
||||
return Ok(AnthropicMessagesRequest {
|
||||
thinking: None,
|
||||
output_config: None,
|
||||
..request
|
||||
});
|
||||
};
|
||||
let Some(fitted) = cap_thinking_budget_to_max_tokens(mapped, request.max_tokens) else {
|
||||
return Ok(request);
|
||||
};
|
||||
let thinking = Some(request.thinking.clone().unwrap_or(fitted));
|
||||
if !context.capabilities.supports_adaptive_thinking {
|
||||
return Ok(AnthropicMessagesRequest {
|
||||
thinking,
|
||||
..request
|
||||
});
|
||||
}
|
||||
let effort = reasoning_effort_to_output_config_effort(&reasoning_effort).ok_or_else(|| {
|
||||
bad_request(format!(
|
||||
"Invalid reasoning_effort: {reasoning_effort:?}. Must be one of: {EFFORT_NAMES}"
|
||||
))
|
||||
})?;
|
||||
if let Some(rejection) = context
|
||||
.capabilities
|
||||
.effort_level_rejection(effort, &request.model)
|
||||
{
|
||||
return Err(bad_request(rejection));
|
||||
}
|
||||
Ok(AnthropicMessagesRequest {
|
||||
thinking,
|
||||
output_config: Some(with_effort(request.output_config.clone(), effort, true)),
|
||||
..request
|
||||
})
|
||||
}
|
||||
|
||||
/// Always-on-thinking models 400 on `thinking.type=disabled`; omit it instead.
|
||||
fn drop_disabled_thinking(
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &ThinkingContext,
|
||||
) -> AnthropicMessagesRequest {
|
||||
if !context.capabilities.thinking_always_on
|
||||
|| thinking_type(request.thinking.as_ref()) != Some("disabled")
|
||||
{
|
||||
return request;
|
||||
}
|
||||
AnthropicMessagesRequest {
|
||||
thinking: None,
|
||||
..request
|
||||
}
|
||||
}
|
||||
|
||||
/// Adaptive models that reject the legacy shape get `thinking.type=enabled` translated to
|
||||
/// adaptive plus an `output_config.effort` bucketed from the budget.
|
||||
fn translate_legacy_thinking_for_adaptive_model(
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &ThinkingContext,
|
||||
) -> AnthropicMessagesRequest {
|
||||
let capabilities = &context.capabilities;
|
||||
if !capabilities.supports_adaptive_thinking
|
||||
|| capabilities.supports_legacy_thinking
|
||||
|| thinking_type(request.thinking.as_ref()) != Some("enabled")
|
||||
{
|
||||
return request;
|
||||
}
|
||||
let budget = request
|
||||
.thinking
|
||||
.as_ref()
|
||||
.and_then(|thinking| thinking.get("budget_tokens"))
|
||||
.and_then(Value::as_u64)
|
||||
.unwrap_or(0);
|
||||
let effort = context.budgets.effort_for_budget(budget, capabilities);
|
||||
AnthropicMessagesRequest {
|
||||
thinking: Some(json!({"type": "adaptive"})),
|
||||
output_config: Some(with_effort(request.output_config.clone(), effort, true)),
|
||||
..request
|
||||
}
|
||||
}
|
||||
|
||||
fn output_config_without_effort(output_config: Option<Value>) -> Option<Value> {
|
||||
let Some(Value::Object(config)) = output_config else {
|
||||
return output_config;
|
||||
};
|
||||
if !config.contains_key("effort") {
|
||||
return Some(Value::Object(config));
|
||||
}
|
||||
let residual: Map<String, Value> = config
|
||||
.into_iter()
|
||||
.filter(|(key, _)| key != "effort")
|
||||
.collect();
|
||||
(!residual.is_empty()).then_some(Value::Object(residual))
|
||||
}
|
||||
|
||||
/// The 4.6+ adaptive interface (`thinking.type=adaptive`, `output_config.effort`) reshaped
|
||||
/// for an older model: native effort is kept where the model takes it, otherwise the effort
|
||||
/// becomes a capped legacy budget, or thinking is dropped for models without reasoning.
|
||||
fn translate_adaptive_effort_for_non_adaptive_model(
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &ThinkingContext,
|
||||
) -> Result<AnthropicMessagesRequest, Error> {
|
||||
let capabilities = &context.capabilities;
|
||||
if capabilities.supports_adaptive_thinking {
|
||||
return Ok(request);
|
||||
}
|
||||
let effort = output_config_effort(request.output_config.as_ref()).map(str::to_string);
|
||||
let adaptive_thinking = thinking_type(request.thinking.as_ref()) == Some("adaptive");
|
||||
if effort.is_none() && !adaptive_thinking {
|
||||
return Ok(request);
|
||||
}
|
||||
let level_supported = effort.as_deref().is_none_or(|effort| {
|
||||
capabilities
|
||||
.effort_level_rejection(effort, &request.model)
|
||||
.is_none()
|
||||
});
|
||||
if capabilities.supports_effort_param() && (!adaptive_thinking || level_supported) {
|
||||
return Ok(AnthropicMessagesRequest {
|
||||
thinking: if adaptive_thinking {
|
||||
None
|
||||
} else {
|
||||
request.thinking.clone()
|
||||
},
|
||||
..request
|
||||
});
|
||||
}
|
||||
let legacy = if capabilities.supports_reasoning {
|
||||
map_reasoning_effort(effort.as_deref().unwrap_or("medium"), context)?
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let capped =
|
||||
legacy.and_then(|thinking| cap_thinking_budget_to_max_tokens(thinking, request.max_tokens));
|
||||
Ok(AnthropicMessagesRequest {
|
||||
thinking: capped,
|
||||
output_config: output_config_without_effort(request.output_config.clone()),
|
||||
..request
|
||||
})
|
||||
}
|
||||
|
||||
/// Anthropic only accepts `temperature=1` while extended thinking is on, so a pinned
|
||||
/// temperature is dropped on non-adaptive models once thinking or effort is in play.
|
||||
fn drop_incompatible_temperature_for_thinking(
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &ThinkingContext,
|
||||
) -> AnthropicMessagesRequest {
|
||||
if context.capabilities.supports_adaptive_thinking {
|
||||
return request;
|
||||
}
|
||||
let pinned = request
|
||||
.temperature
|
||||
.is_some_and(|temperature| temperature != 1.0);
|
||||
let thinking_enabled = thinking_type(request.thinking.as_ref()) == Some("enabled");
|
||||
let effort_enabled = output_config_effort(request.output_config.as_ref()).is_some();
|
||||
if !pinned || !(thinking_enabled || effort_enabled) {
|
||||
return request;
|
||||
}
|
||||
AnthropicMessagesRequest {
|
||||
temperature: None,
|
||||
..request
|
||||
}
|
||||
}
|
||||
|
||||
/// The full reasoning translation, in Python's order.
|
||||
pub fn translate_thinking(
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &ThinkingContext,
|
||||
) -> Result<AnthropicMessagesRequest, Error> {
|
||||
let request = translate_reasoning_effort(request, context)?;
|
||||
let request = drop_disabled_thinking(request, context);
|
||||
let request = translate_legacy_thinking_for_adaptive_model(request, context);
|
||||
let request = translate_adaptive_effort_for_non_adaptive_model(request, context)?;
|
||||
Ok(drop_incompatible_temperature_for_thinking(request, context))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use rstest::rstest;
|
||||
|
||||
use super::*;
|
||||
use crate::anthropic::common_utils::SupportedEffortTiers;
|
||||
|
||||
fn request(fields: Value) -> AnthropicMessagesRequest {
|
||||
let mut body =
|
||||
json!({"model": "claude", "messages": [{"role": "user", "content": "Hello"}]});
|
||||
body.as_object_mut()
|
||||
.unwrap()
|
||||
.extend(fields.as_object().unwrap().clone());
|
||||
serde_json::from_value(body).unwrap()
|
||||
}
|
||||
|
||||
fn body(request: &AnthropicMessagesRequest) -> Value {
|
||||
serde_json::to_value(request).unwrap()
|
||||
}
|
||||
|
||||
fn context(capabilities: AnthropicModelCapabilities) -> ThinkingContext {
|
||||
ThinkingContext {
|
||||
capabilities,
|
||||
budgets: ThinkingBudgets::default(),
|
||||
}
|
||||
}
|
||||
|
||||
fn haiku_4_5() -> AnthropicModelCapabilities {
|
||||
AnthropicModelCapabilities {
|
||||
supports_reasoning: true,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn opus_4_5() -> AnthropicModelCapabilities {
|
||||
AnthropicModelCapabilities {
|
||||
supports_reasoning: true,
|
||||
supports_output_config: true,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn sonnet_4_6() -> AnthropicModelCapabilities {
|
||||
AnthropicModelCapabilities {
|
||||
supports_reasoning: true,
|
||||
supports_adaptive_thinking: true,
|
||||
supports_legacy_thinking: true,
|
||||
supports_output_config: true,
|
||||
effort_tiers: SupportedEffortTiers {
|
||||
max: true,
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn opus_4_7() -> AnthropicModelCapabilities {
|
||||
AnthropicModelCapabilities {
|
||||
supports_reasoning: true,
|
||||
supports_adaptive_thinking: true,
|
||||
supports_output_config: true,
|
||||
effort_tiers: SupportedEffortTiers {
|
||||
xhigh: true,
|
||||
max: true,
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn fable_5_1() -> AnthropicModelCapabilities {
|
||||
AnthropicModelCapabilities {
|
||||
thinking_always_on: true,
|
||||
..opus_4_7()
|
||||
}
|
||||
}
|
||||
|
||||
fn claude_code_payload(effort: &str, max_tokens: u64) -> Value {
|
||||
json!({"max_tokens": max_tokens, "thinking": {"type": "adaptive"}, "output_config": {"effort": effort}})
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case("minimal", "low")]
|
||||
#[case("low", "low")]
|
||||
#[case("medium", "medium")]
|
||||
#[case("high", "high")]
|
||||
#[case("xhigh", "xhigh")]
|
||||
#[case("max", "max")]
|
||||
fn reasoning_effort_maps_to_output_config_for_adaptive_model(
|
||||
#[case] reasoning_effort: &str,
|
||||
#[case] expected_effort: &str,
|
||||
) {
|
||||
let result = translate_thinking(
|
||||
request(json!({"max_tokens": 1024, "reasoning_effort": reasoning_effort})),
|
||||
&context(opus_4_7()),
|
||||
)
|
||||
.unwrap();
|
||||
let sent = body(&result);
|
||||
assert!(sent.get("reasoning_effort").is_none());
|
||||
assert_eq!(
|
||||
sent["thinking"],
|
||||
json!({"type": "adaptive", "display": "summarized"})
|
||||
);
|
||||
assert_eq!(sent["output_config"], json!({"effort": expected_effort}));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reasoning_effort_none_clears_thinking_and_output_config() {
|
||||
let result = translate_thinking(
|
||||
request(json!({
|
||||
"max_tokens": 1024,
|
||||
"reasoning_effort": "none",
|
||||
"thinking": {"type": "adaptive"},
|
||||
"output_config": {"effort": "high"}
|
||||
})),
|
||||
&context(opus_4_7()),
|
||||
)
|
||||
.unwrap();
|
||||
let sent = body(&result);
|
||||
assert!(sent.get("thinking").is_none());
|
||||
assert!(sent.get("output_config").is_none());
|
||||
assert!(sent.get("reasoning_effort").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reasoning_effort_on_non_adaptive_model_uses_a_capped_budget() {
|
||||
let result = translate_thinking(
|
||||
request(json!({"max_tokens": 8192, "reasoning_effort": "high"})),
|
||||
&context(opus_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
result.thinking,
|
||||
Some(json!({"type": "enabled", "budget_tokens": 4096}))
|
||||
);
|
||||
assert!(result.output_config.is_none());
|
||||
|
||||
let capped = translate_thinking(
|
||||
request(json!({"max_tokens": 3000, "reasoning_effort": "high"})),
|
||||
&context(haiku_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
capped.thinking,
|
||||
Some(json!({"type": "enabled", "budget_tokens": 2999}))
|
||||
);
|
||||
|
||||
let dropped = translate_thinking(
|
||||
request(json!({"max_tokens": 512, "reasoning_effort": "high"})),
|
||||
&context(haiku_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(dropped.thinking.is_none());
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case("bogus")]
|
||||
#[case("ultra")]
|
||||
fn invalid_reasoning_effort_is_a_request_error(#[case] bad_effort: &str) {
|
||||
let error = translate_thinking(
|
||||
request(json!({"max_tokens": 1024, "reasoning_effort": bad_effort})),
|
||||
&context(opus_4_5()),
|
||||
)
|
||||
.expect_err("rejected");
|
||||
assert!(
|
||||
matches!(error, Error::InvalidRequest(message) if message.contains("Unmapped reasoning effort"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reasoning_effort_unsupported_tier_is_a_request_error_on_adaptive_models() {
|
||||
let error = translate_thinking(
|
||||
request(json!({"max_tokens": 1024, "reasoning_effort": "xhigh"})),
|
||||
&context(sonnet_4_6()),
|
||||
)
|
||||
.expect_err("rejected");
|
||||
assert!(
|
||||
matches!(error, Error::InvalidRequest(message) if message.contains("effort='xhigh'"))
|
||||
);
|
||||
let accepted = translate_thinking(
|
||||
request(json!({"max_tokens": 1024, "reasoning_effort": "max"})),
|
||||
&context(sonnet_4_6()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(accepted.output_config, Some(json!({"effort": "max"})));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn explicit_thinking_and_output_config_win_over_reasoning_effort() {
|
||||
let result = translate_thinking(
|
||||
request(json!({
|
||||
"max_tokens": 16000,
|
||||
"reasoning_effort": "low",
|
||||
"thinking": {"type": "enabled", "budget_tokens": 8000},
|
||||
"output_config": {"effort": "high"}
|
||||
})),
|
||||
&context(sonnet_4_6()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
result.thinking,
|
||||
Some(json!({"type": "enabled", "budget_tokens": 8000}))
|
||||
);
|
||||
assert_eq!(result.output_config, Some(json!({"effort": "high"})));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn legacy_thinking_is_preserved_verbatim_on_4_6() {
|
||||
let result = translate_thinking(
|
||||
request(json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 31999}})),
|
||||
&context(sonnet_4_6()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
result.thinking,
|
||||
Some(json!({"type": "enabled", "budget_tokens": 31999}))
|
||||
);
|
||||
assert!(result.output_config.is_none());
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case(1000, "low")]
|
||||
#[case(2048, "medium")]
|
||||
#[case(4096, "high")]
|
||||
#[case(24000, "xhigh")]
|
||||
fn legacy_thinking_translates_to_adaptive_buckets_on_4_7(
|
||||
#[case] budget_tokens: u64,
|
||||
#[case] expected_effort: &str,
|
||||
) {
|
||||
let result = translate_thinking(
|
||||
request(json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": budget_tokens}})),
|
||||
&context(opus_4_7()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(result.thinking, Some(json!({"type": "adaptive"})));
|
||||
assert_eq!(
|
||||
result.output_config,
|
||||
Some(json!({"effort": expected_effort}))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn legacy_thinking_does_not_override_explicit_output_config() {
|
||||
let result = translate_thinking(
|
||||
request(json!({
|
||||
"max_tokens": 32000,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 31999},
|
||||
"output_config": {"effort": "low", "format": {"type": "json_schema"}}
|
||||
})),
|
||||
&context(opus_4_7()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
result.output_config,
|
||||
Some(json!({"effort": "low", "format": {"type": "json_schema"}}))
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case(fable_5_1(), true)]
|
||||
#[case(opus_4_7(), false)]
|
||||
fn disabled_thinking_is_omitted_only_for_always_on_models(
|
||||
#[case] capabilities: AnthropicModelCapabilities,
|
||||
#[case] expected_dropped: bool,
|
||||
) {
|
||||
let result = translate_thinking(
|
||||
request(json!({"max_tokens": 64, "thinking": {"type": "disabled"}})),
|
||||
&context(capabilities),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(result.thinking.is_none(), expected_dropped);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adaptive_payload_is_translated_to_legacy_for_haiku_4_5() {
|
||||
let result = translate_thinking(
|
||||
request(claude_code_payload("high", 8192)),
|
||||
&context(haiku_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
result.thinking,
|
||||
Some(json!({"type": "enabled", "budget_tokens": 4096}))
|
||||
);
|
||||
assert!(result.output_config.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adaptive_thinking_only_is_translated_or_dropped_by_reasoning_support() {
|
||||
let translated = translate_thinking(
|
||||
request(json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}})),
|
||||
&context(haiku_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
translated.thinking,
|
||||
Some(json!({"type": "enabled", "budget_tokens": 2048}))
|
||||
);
|
||||
|
||||
let dropped = translate_thinking(
|
||||
request(claude_code_payload("medium", 8192)),
|
||||
&context(AnthropicModelCapabilities::default()),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(dropped.thinking.is_none());
|
||||
assert!(dropped.output_config.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adaptive_payload_passes_through_untouched_on_4_6() {
|
||||
let result = translate_thinking(
|
||||
request(claude_code_payload("medium", 8192)),
|
||||
&context(sonnet_4_6()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(result.thinking, Some(json!({"type": "adaptive"})));
|
||||
assert_eq!(result.output_config, Some(json!({"effort": "medium"})));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn residual_output_config_survives_effort_translation() {
|
||||
let result = translate_thinking(
|
||||
request(json!({
|
||||
"max_tokens": 8192,
|
||||
"thinking": {"type": "adaptive"},
|
||||
"output_config": {"effort": "medium", "format": {"type": "json_schema"}}
|
||||
})),
|
||||
&context(haiku_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
result.output_config,
|
||||
Some(json!({"format": {"type": "json_schema"}}))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn opus_4_5_keeps_supported_effort_and_drops_adaptive_thinking() {
|
||||
let result = translate_thinking(
|
||||
request(claude_code_payload("high", 8192)),
|
||||
&context(opus_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(result.thinking.is_none());
|
||||
assert_eq!(result.output_config, Some(json!({"effort": "high"})));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn opus_4_5_unsupported_effort_with_adaptive_thinking_falls_back_to_legacy() {
|
||||
let result = translate_thinking(
|
||||
request(claude_code_payload("xhigh", 64000)),
|
||||
&context(opus_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
result.thinking,
|
||||
Some(json!({"type": "enabled", "budget_tokens": 8192}))
|
||||
);
|
||||
assert!(result.output_config.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn opus_4_5_effort_only_request_is_left_untouched() {
|
||||
let result = translate_thinking(
|
||||
request(json!({"max_tokens": 4096, "output_config": {"effort": "xhigh"}})),
|
||||
&context(opus_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(result.thinking.is_none());
|
||||
assert_eq!(result.output_config, Some(json!({"effort": "xhigh"})));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn budget_is_capped_below_max_tokens_and_dropped_when_it_cannot_fit() {
|
||||
let capped = translate_thinking(
|
||||
request(claude_code_payload("high", 3000)),
|
||||
&context(haiku_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
capped.thinking,
|
||||
Some(json!({"type": "enabled", "budget_tokens": 2999}))
|
||||
);
|
||||
|
||||
let dropped = translate_thinking(
|
||||
request(claude_code_payload("medium", 512)),
|
||||
&context(haiku_4_5()),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(dropped.thinking.is_none());
|
||||
assert!(dropped.output_config.is_none());
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case(haiku_4_5(), claude_code_payload("medium", 8192), 0.0, true)]
|
||||
#[case(haiku_4_5(), claude_code_payload("medium", 8192), 1.0, false)]
|
||||
#[case(haiku_4_5(), claude_code_payload("medium", 512), 0.0, false)]
|
||||
#[case(opus_4_7(), claude_code_payload("medium", 8192), 0.0, false)]
|
||||
#[case(opus_4_5(), claude_code_payload("high", 8192), 0.0, true)]
|
||||
#[case(haiku_4_5(), json!({"max_tokens": 8192, "reasoning_effort": "high"}), 0.2, true)]
|
||||
#[case(haiku_4_5(), json!({"max_tokens": 8192}), 0.0, false)]
|
||||
fn pinned_temperature_is_dropped_only_when_thinking_survives_on_a_non_adaptive_model(
|
||||
#[case] capabilities: AnthropicModelCapabilities,
|
||||
#[case] payload: Value,
|
||||
#[case] temperature: f64,
|
||||
#[case] expected_dropped: bool,
|
||||
) {
|
||||
let mut payload = payload;
|
||||
payload
|
||||
.as_object_mut()
|
||||
.unwrap()
|
||||
.insert("temperature".to_string(), json!(temperature));
|
||||
let result = translate_thinking(request(payload), &context(capabilities)).unwrap();
|
||||
assert_eq!(result.temperature.is_none(), expected_dropped);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn budgets_honor_environment_overrides() {
|
||||
let env = |name: &str| {
|
||||
(name == "DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET").then(|| " 6000 ".to_string())
|
||||
};
|
||||
let budgets = ThinkingBudgets::from_lookup(&env);
|
||||
assert_eq!(budgets.high, 6000);
|
||||
assert_eq!(budgets.medium, ThinkingBudgets::default().medium);
|
||||
}
|
||||
}
|
||||
|
|
@ -1,9 +1,27 @@
|
|||
use crate::base_llm::{
|
||||
anthropic_messages::transformation::BaseAnthropicMessagesConfig, chat::transformation::Error,
|
||||
use litellm_core_utils::settings::{Lookup, ProcessEnvironment};
|
||||
use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest;
|
||||
use serde_json::{Map, Value, json};
|
||||
|
||||
use super::{
|
||||
headers::{Headers, authenticate, with_feature_betas},
|
||||
thinking::{ThinkingBudgets, ThinkingContext, translate_thinking},
|
||||
};
|
||||
use crate::{
|
||||
anthropic::common_utils::{
|
||||
AnthropicModelCapabilities, has_advisor_tool, strip_advisor_blocks,
|
||||
strip_encrypted_reasoning_blocks,
|
||||
},
|
||||
base_llm::{
|
||||
anthropic_messages::transformation::{
|
||||
BaseAnthropicMessagesConfig, MessagesTransformContext,
|
||||
},
|
||||
chat::transformation::Error,
|
||||
},
|
||||
};
|
||||
|
||||
const ANTHROPIC_API_KEY_ENV: &str = "ANTHROPIC_API_KEY";
|
||||
const ANTHROPIC_API_BASE_ENV: &str = "ANTHROPIC_API_BASE";
|
||||
const ANTHROPIC_BASE_URL_ENV: &str = "ANTHROPIC_BASE_URL";
|
||||
const DEFAULT_ANTHROPIC_API_BASE: &str = "https://api.anthropic.com";
|
||||
const MESSAGES_PATH_SUFFIX: &str = "/v1/messages";
|
||||
|
||||
|
|
@ -11,6 +29,28 @@ pub struct AnthropicMessagesConfig;
|
|||
|
||||
pub const ANTHROPIC_MESSAGES_CONFIG: AnthropicMessagesConfig = AnthropicMessagesConfig;
|
||||
|
||||
impl MessagesTransformContext {
|
||||
/// The context for a model whose capability flags the host resolved, with the thinking
|
||||
/// budgets read from the process environment.
|
||||
pub fn new(capabilities: AnthropicModelCapabilities, drop_params: bool) -> Self {
|
||||
Self::with_lookup(capabilities, drop_params, &ProcessEnvironment)
|
||||
}
|
||||
|
||||
pub fn with_lookup(
|
||||
capabilities: AnthropicModelCapabilities,
|
||||
drop_params: bool,
|
||||
env: &impl Lookup,
|
||||
) -> Self {
|
||||
Self {
|
||||
thinking: ThinkingContext {
|
||||
capabilities,
|
||||
budgets: ThinkingBudgets::from_lookup(env),
|
||||
},
|
||||
drop_params,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig {
|
||||
fn get_complete_url(
|
||||
&self,
|
||||
|
|
@ -21,6 +61,35 @@ impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig {
|
|||
Ok(complete_anthropic_url(api_base, env_lookup))
|
||||
}
|
||||
|
||||
fn transform_anthropic_messages_request(
|
||||
&self,
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &MessagesTransformContext,
|
||||
) -> Result<AnthropicMessagesRequest, Error> {
|
||||
if request.max_tokens.is_none() {
|
||||
return Err(Error::InvalidRequest(
|
||||
"max_tokens is required for Anthropic /v1/messages API".to_string(),
|
||||
));
|
||||
}
|
||||
let request = drop_unsupported_params(request, context)?;
|
||||
let request = translate_thinking(request, &context.thinking)?;
|
||||
let context_management = request
|
||||
.context_management
|
||||
.as_ref()
|
||||
.and_then(map_openai_context_management_to_anthropic)
|
||||
.or_else(|| request.context_management.clone());
|
||||
let messages = if has_advisor_tool(request.tools.as_deref()) {
|
||||
request.messages
|
||||
} else {
|
||||
strip_advisor_blocks(request.messages)
|
||||
};
|
||||
Ok(AnthropicMessagesRequest {
|
||||
messages: strip_encrypted_reasoning_blocks(messages),
|
||||
context_management,
|
||||
..request
|
||||
})
|
||||
}
|
||||
|
||||
fn resolve_api_key(
|
||||
&self,
|
||||
api_key: Option<&str>,
|
||||
|
|
@ -28,6 +97,109 @@ impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig {
|
|||
) -> Result<String, Error> {
|
||||
resolve_anthropic_api_key(api_key, env_lookup).map_err(Error::from)
|
||||
}
|
||||
|
||||
fn authenticate(
|
||||
&self,
|
||||
headers: Headers,
|
||||
api_key: Option<&str>,
|
||||
env_lookup: &dyn Fn(&str) -> Option<String>,
|
||||
) -> Result<Headers, Error> {
|
||||
authenticate(headers, api_key, env_lookup).map_err(Error::from)
|
||||
}
|
||||
|
||||
fn request_headers(&self, headers: Headers, request: &AnthropicMessagesRequest) -> Headers {
|
||||
with_feature_betas(headers, request)
|
||||
}
|
||||
}
|
||||
|
||||
fn unsupported_param(model: &str, param: &str, value: &Value, hint: &str) -> Error {
|
||||
Error::InvalidRequest(format!(
|
||||
"{model} does not support {param}={value}. {hint}To drop unsupported params, set `litellm.drop_params = True`."
|
||||
))
|
||||
}
|
||||
|
||||
/// Python's `_maybe_drop_speed_param` and `_apply_sampling_param` gates: a parameter the
|
||||
/// model removed is dropped under `drop_params`, else rejected before the call.
|
||||
fn drop_unsupported_params(
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &MessagesTransformContext,
|
||||
) -> Result<AnthropicMessagesRequest, Error> {
|
||||
let capabilities = &context.thinking.capabilities;
|
||||
let model = request.model.clone();
|
||||
let reject = |param: &str, value: Value, hint: &str| -> Result<(), Error> {
|
||||
if context.drop_params {
|
||||
return Ok(());
|
||||
}
|
||||
Err(unsupported_param(&model, param, &value, hint))
|
||||
};
|
||||
let speed = match request.speed.as_deref() {
|
||||
Some(speed) if !capabilities.supports_speed => {
|
||||
reject("speed", json!(speed), "")?;
|
||||
None
|
||||
}
|
||||
_ => request.speed.clone(),
|
||||
};
|
||||
if capabilities.supports_sampling_params {
|
||||
return Ok(AnthropicMessagesRequest { speed, ..request });
|
||||
}
|
||||
let temperature = match request.temperature {
|
||||
Some(temperature) if temperature != 1.0 => {
|
||||
reject(
|
||||
"temperature",
|
||||
json!(temperature),
|
||||
"Only temperature=1 is supported. ",
|
||||
)?;
|
||||
None
|
||||
}
|
||||
temperature => temperature,
|
||||
};
|
||||
if let Some(top_p) = request.top_p {
|
||||
reject("top_p", json!(top_p), "")?;
|
||||
}
|
||||
if let Some(top_k) = request.top_k {
|
||||
reject("top_k", json!(top_k), "")?;
|
||||
}
|
||||
Ok(AnthropicMessagesRequest {
|
||||
speed,
|
||||
temperature,
|
||||
top_p: None,
|
||||
top_k: None,
|
||||
..request
|
||||
})
|
||||
}
|
||||
|
||||
/// OpenAI's `[{"type": "compaction", "compact_threshold": N}]` list becomes Anthropic's
|
||||
/// `{"edits": [{"type": "compact_20260112", "trigger": {...}}]}`. An Anthropic-shaped value
|
||||
/// is returned as is; anything else yields `None`.
|
||||
pub fn map_openai_context_management_to_anthropic(context_management: &Value) -> Option<Value> {
|
||||
match context_management {
|
||||
Value::Object(edits) if edits.contains_key("edits") => Some(context_management.clone()),
|
||||
Value::Array(entries) => {
|
||||
let edits: Vec<Value> = entries
|
||||
.iter()
|
||||
.filter_map(Value::as_object)
|
||||
.filter(|entry| entry.get("type").and_then(Value::as_str) == Some("compaction"))
|
||||
.map(|entry| {
|
||||
let trigger = entry.get("compact_threshold").and_then(Value::as_f64).map(
|
||||
|threshold| json!({"type": "input_tokens", "value": threshold as i64}),
|
||||
);
|
||||
let passthrough = entry
|
||||
.iter()
|
||||
.filter(|(key, _)| !matches!(key.as_str(), "type" | "compact_threshold"))
|
||||
.map(|(key, value)| (key.clone(), value.clone()));
|
||||
Value::Object(
|
||||
[("type".to_string(), json!("compact_20260112"))]
|
||||
.into_iter()
|
||||
.chain(trigger.map(|trigger| ("trigger".to_string(), trigger)))
|
||||
.chain(passthrough)
|
||||
.collect::<Map<String, Value>>(),
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
(!edits.is_empty()).then(|| json!({"edits": edits}))
|
||||
}
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn non_empty(value: Option<&str>) -> Option<&str> {
|
||||
|
|
@ -60,20 +232,58 @@ pub fn complete_anthropic_url(
|
|||
format!("{api_base}{MESSAGES_PATH_SUFFIX}")
|
||||
}
|
||||
|
||||
/// `api_base`, else `ANTHROPIC_API_BASE`, else the SDK's `ANTHROPIC_BASE_URL`, else the
|
||||
/// public endpoint.
|
||||
pub fn resolve_anthropic_api_base(
|
||||
api_base: Option<&str>,
|
||||
env_lookup: &dyn Fn(&str) -> Option<String>,
|
||||
) -> String {
|
||||
let env = |name: &str| env_lookup(name).filter(|value| !value.trim().is_empty());
|
||||
non_empty(api_base)
|
||||
.map(str::to_string)
|
||||
.or_else(|| env_lookup(ANTHROPIC_API_BASE_ENV).filter(|value| !value.trim().is_empty()))
|
||||
.or_else(|| env(ANTHROPIC_API_BASE_ENV))
|
||||
.or_else(|| env(ANTHROPIC_BASE_URL_ENV))
|
||||
.unwrap_or_else(|| DEFAULT_ANTHROPIC_API_BASE.to_string())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use rstest::rstest;
|
||||
|
||||
use super::*;
|
||||
|
||||
fn request(fields: Value) -> AnthropicMessagesRequest {
|
||||
let mut body = json!({
|
||||
"model": "claude",
|
||||
"max_tokens": 1024,
|
||||
"messages": [{"role": "user", "content": "Hello"}]
|
||||
});
|
||||
body.as_object_mut()
|
||||
.unwrap()
|
||||
.extend(fields.as_object().unwrap().clone());
|
||||
serde_json::from_value(body).unwrap()
|
||||
}
|
||||
|
||||
fn context(
|
||||
capabilities: AnthropicModelCapabilities,
|
||||
drop_params: bool,
|
||||
) -> MessagesTransformContext {
|
||||
MessagesTransformContext::with_lookup(capabilities, drop_params, &|_: &str| None)
|
||||
}
|
||||
|
||||
fn transform(
|
||||
fields: Value,
|
||||
capabilities: AnthropicModelCapabilities,
|
||||
drop_params: bool,
|
||||
) -> Result<Value, Error> {
|
||||
ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(
|
||||
request(fields),
|
||||
&context(capabilities, drop_params),
|
||||
)
|
||||
.map(|transformed| serde_json::to_value(transformed).unwrap())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn url_defaults_to_public_anthropic_endpoint() {
|
||||
assert_eq!(
|
||||
|
|
@ -98,11 +308,11 @@ mod tests {
|
|||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn url_falls_back_to_env_base() {
|
||||
let with_env = |key: &str| {
|
||||
(key == ANTHROPIC_API_BASE_ENV).then(|| "https://env.anthropic".to_string())
|
||||
};
|
||||
#[rstest]
|
||||
#[case(ANTHROPIC_API_BASE_ENV)]
|
||||
#[case(ANTHROPIC_BASE_URL_ENV)]
|
||||
fn url_falls_back_to_either_env_base(#[case] variable: &str) {
|
||||
let with_env = |key: &str| (key == variable).then(|| "https://env.anthropic".to_string());
|
||||
assert_eq!(
|
||||
complete_anthropic_url(Some(" "), &with_env),
|
||||
"https://env.anthropic/v1/messages"
|
||||
|
|
@ -142,4 +352,161 @@ mod tests {
|
|||
]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn max_tokens_is_required() {
|
||||
let mut body = request(json!({}));
|
||||
body.max_tokens = None;
|
||||
let error = ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(
|
||||
body,
|
||||
&context(AnthropicModelCapabilities::default(), false),
|
||||
)
|
||||
.expect_err("rejected");
|
||||
assert!(
|
||||
matches!(error, Error::InvalidRequest(message) if message.contains("max_tokens is required"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn advisor_history_is_stripped_only_when_the_advisor_tool_is_absent() {
|
||||
let history = json!([
|
||||
{"role": "user", "content": "q"},
|
||||
{"role": "assistant", "content": [
|
||||
{"type": "server_tool_use", "id": "adv", "name": "advisor", "input": {}},
|
||||
{"type": "advisor_tool_result", "tool_use_id": "adv", "content": "x"},
|
||||
{"type": "text", "text": "a"}
|
||||
]}
|
||||
]);
|
||||
let stripped = transform(
|
||||
json!({"messages": history}),
|
||||
AnthropicModelCapabilities::default(),
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
stripped["messages"][1]["content"],
|
||||
json!([{"type": "text", "text": "a"}])
|
||||
);
|
||||
let kept = transform(
|
||||
json!({"messages": history, "tools": [{"type": "advisor_20260301", "name": "advisor"}]}),
|
||||
AnthropicModelCapabilities::default(),
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(kept["messages"][1]["content"].as_array().unwrap().len(), 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encrypted_reasoning_blocks_are_stripped_from_the_wire() {
|
||||
let sent = transform(
|
||||
json!({"messages": [
|
||||
{"role": "user", "content": "q"},
|
||||
{"role": "assistant", "content": [
|
||||
{"type": "thinking", "thinking": "t", "signature": "litellm_encrypted_reasoning:abc"},
|
||||
{"type": "text", "text": "a"}
|
||||
]}
|
||||
]}),
|
||||
AnthropicModelCapabilities::default(),
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
sent["messages"][1]["content"],
|
||||
json!([{"type": "text", "text": "a"}])
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn openai_context_management_list_becomes_anthropic_edits() {
|
||||
let sent = transform(
|
||||
json!({"context_management": [
|
||||
{"type": "compaction", "compact_threshold": 150000, "note": "keep"},
|
||||
{"type": "other"}
|
||||
]}),
|
||||
AnthropicModelCapabilities::default(),
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
sent["context_management"],
|
||||
json!({"edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}, "note": "keep"}]})
|
||||
);
|
||||
let native = json!({"edits": [{"type": "clear_tool_uses_20250919"}]});
|
||||
let untouched = transform(
|
||||
json!({"context_management": native}),
|
||||
AnthropicModelCapabilities::default(),
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(untouched["context_management"], native);
|
||||
assert_eq!(
|
||||
map_openai_context_management_to_anthropic(&json!([{"type": "other"}])),
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
fn no_sampling() -> AnthropicModelCapabilities {
|
||||
AnthropicModelCapabilities {
|
||||
supports_sampling_params: false,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn removed_sampling_params_are_dropped_under_drop_params_and_rejected_otherwise() {
|
||||
let dropped = transform(
|
||||
json!({"temperature": 0.2, "top_p": 0.9, "top_k": 5}),
|
||||
no_sampling(),
|
||||
true,
|
||||
)
|
||||
.unwrap();
|
||||
for param in ["temperature", "top_p", "top_k"] {
|
||||
assert!(dropped.get(param).is_none(), "{param} should be dropped");
|
||||
}
|
||||
let kept_unit_temperature =
|
||||
transform(json!({"temperature": 1.0}), no_sampling(), false).unwrap();
|
||||
assert_eq!(kept_unit_temperature["temperature"], json!(1.0));
|
||||
let rejected = transform(json!({"top_k": 5}), no_sampling(), false).expect_err("rejected");
|
||||
assert!(
|
||||
matches!(rejected, Error::InvalidRequest(message) if message.contains("does not support top_k=5"))
|
||||
);
|
||||
let supported = transform(
|
||||
json!({"temperature": 0.2, "top_p": 0.9, "top_k": 5}),
|
||||
AnthropicModelCapabilities::default(),
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(supported["top_k"], json!(5));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn speed_is_gated_on_the_model_flag() {
|
||||
let dropped = transform(
|
||||
json!({"speed": "fast"}),
|
||||
AnthropicModelCapabilities::default(),
|
||||
true,
|
||||
)
|
||||
.unwrap();
|
||||
assert!(dropped.get("speed").is_none());
|
||||
let rejected = transform(
|
||||
json!({"speed": "fast"}),
|
||||
AnthropicModelCapabilities::default(),
|
||||
false,
|
||||
)
|
||||
.expect_err("rejected");
|
||||
assert!(
|
||||
matches!(rejected, Error::InvalidRequest(message) if message.contains("does not support speed="))
|
||||
);
|
||||
let kept = transform(
|
||||
json!({"speed": "fast"}),
|
||||
AnthropicModelCapabilities {
|
||||
supports_speed: true,
|
||||
..Default::default()
|
||||
},
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(kept["speed"], json!("fast"));
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
pub mod batches;
|
||||
pub mod chat;
|
||||
pub mod common_utils;
|
||||
pub mod count_tokens;
|
||||
pub mod experimental_pass_through;
|
||||
|
||||
|
|
|
|||
|
|
@ -4,14 +4,15 @@ use litellm_types::llms::anthropic_messages::{
|
|||
},
|
||||
anthropic_response::AnthropicMessagesResponse,
|
||||
};
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
use crate::{
|
||||
anthropic::experimental_pass_through::messages::transformation::{
|
||||
ANTHROPIC_MESSAGES_CONFIG, AnthropicMessagesConfig, non_empty,
|
||||
},
|
||||
base_llm::{
|
||||
anthropic_messages::transformation::{BaseAnthropicMessagesConfig, MessagesAuthStrategy},
|
||||
anthropic_messages::transformation::{
|
||||
BaseAnthropicMessagesConfig, MessagesAuthStrategy, MessagesTransformContext,
|
||||
},
|
||||
chat::transformation::Error,
|
||||
},
|
||||
};
|
||||
|
|
@ -21,7 +22,6 @@ const AZURE_API_BASE_ENV: &str = "AZURE_API_BASE";
|
|||
const ANTHROPIC_PATH_SEGMENT: &str = "/anthropic";
|
||||
const MESSAGES_PATH_SUFFIX: &str = "/v1/messages";
|
||||
const SYSTEM_ROLE: &str = "system";
|
||||
const TEXT_BLOCK_TYPE: &str = "text";
|
||||
|
||||
pub struct AzureAnthropicMessagesConfig {
|
||||
anthropic: AnthropicMessagesConfig,
|
||||
|
|
@ -45,6 +45,7 @@ impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig {
|
|||
fn transform_anthropic_messages_request(
|
||||
&self,
|
||||
request: AnthropicMessagesRequest,
|
||||
context: &MessagesTransformContext,
|
||||
) -> Result<AnthropicMessagesRequest, Error> {
|
||||
let mut request = fold_system_role_messages(request);
|
||||
if let Some(system) = request.system.as_mut() {
|
||||
|
|
@ -54,7 +55,8 @@ impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig {
|
|||
.messages
|
||||
.iter_mut()
|
||||
.for_each(strip_scope_from_message);
|
||||
self.anthropic.transform_anthropic_messages_request(request)
|
||||
self.anthropic
|
||||
.transform_anthropic_messages_request(request, context)
|
||||
}
|
||||
|
||||
fn transform_anthropic_messages_response(
|
||||
|
|
@ -143,17 +145,7 @@ fn strip_scope_from_message(message: &mut AnthropicMessage) {
|
|||
}
|
||||
|
||||
fn text_content_block(text: String) -> ContentBlock {
|
||||
let extra = Map::from_iter([
|
||||
(
|
||||
"type".to_string(),
|
||||
Value::String(TEXT_BLOCK_TYPE.to_string()),
|
||||
),
|
||||
("text".to_string(), Value::String(text)),
|
||||
]);
|
||||
ContentBlock {
|
||||
cache_control: None,
|
||||
extra,
|
||||
}
|
||||
ContentBlock::text(text)
|
||||
}
|
||||
|
||||
fn content_into_blocks(content: MessageContent) -> Vec<ContentBlock> {
|
||||
|
|
@ -202,6 +194,7 @@ mod tests {
|
|||
use serde_json::json;
|
||||
|
||||
use super::*;
|
||||
use crate::anthropic::common_utils::AnthropicModelCapabilities;
|
||||
|
||||
fn request_from(value: serde_json::Value) -> AnthropicMessagesRequest {
|
||||
serde_json::from_value(value).expect("valid request")
|
||||
|
|
@ -346,7 +339,7 @@ mod tests {
|
|||
|
||||
let transformed = to_value(
|
||||
AZURE_ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(request)
|
||||
.transform_anthropic_messages_request(request, &MessagesTransformContext::default())
|
||||
.expect("request transforms"),
|
||||
);
|
||||
|
||||
|
|
@ -373,10 +366,13 @@ mod tests {
|
|||
"messages": [{"role": "user", "content": "hi"}]
|
||||
}));
|
||||
let once = AZURE_ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(request)
|
||||
.transform_anthropic_messages_request(request, &MessagesTransformContext::default())
|
||||
.expect("request transforms");
|
||||
let twice = AZURE_ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(once.clone())
|
||||
.transform_anthropic_messages_request(
|
||||
once.clone(),
|
||||
&MessagesTransformContext::default(),
|
||||
)
|
||||
.expect("request transforms");
|
||||
assert_eq!(once, twice);
|
||||
assert_eq!(to_value(once)["system"], json!("plain string system"));
|
||||
|
|
@ -408,9 +404,21 @@ mod tests {
|
|||
"inference_geo": "us",
|
||||
"litellm_metadata": {"trace": "abc"}
|
||||
});
|
||||
let context = MessagesTransformContext::with_lookup(
|
||||
AnthropicModelCapabilities {
|
||||
supports_reasoning: true,
|
||||
supports_adaptive_thinking: true,
|
||||
supports_legacy_thinking: true,
|
||||
supports_output_config: true,
|
||||
supports_speed: true,
|
||||
..Default::default()
|
||||
},
|
||||
false,
|
||||
&|_: &str| None,
|
||||
);
|
||||
let transformed = to_value(
|
||||
AZURE_ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(request_from(body.clone()))
|
||||
.transform_anthropic_messages_request(request_from(body.clone()), &context)
|
||||
.expect("request transforms"),
|
||||
);
|
||||
assert_eq!(transformed, body);
|
||||
|
|
@ -430,7 +438,7 @@ mod tests {
|
|||
|
||||
let transformed = to_value(
|
||||
AZURE_ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(request)
|
||||
.transform_anthropic_messages_request(request, &MessagesTransformContext::default())
|
||||
.expect("request transforms"),
|
||||
);
|
||||
|
||||
|
|
@ -460,7 +468,7 @@ mod tests {
|
|||
|
||||
let transformed = to_value(
|
||||
AZURE_ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(request)
|
||||
.transform_anthropic_messages_request(request, &MessagesTransformContext::default())
|
||||
.expect("request transforms"),
|
||||
);
|
||||
|
||||
|
|
@ -485,9 +493,21 @@ mod tests {
|
|||
{"role": "assistant", "content": "hello"}
|
||||
]
|
||||
});
|
||||
let context = MessagesTransformContext::with_lookup(
|
||||
AnthropicModelCapabilities {
|
||||
supports_reasoning: true,
|
||||
supports_adaptive_thinking: true,
|
||||
supports_legacy_thinking: true,
|
||||
supports_output_config: true,
|
||||
supports_speed: true,
|
||||
..Default::default()
|
||||
},
|
||||
false,
|
||||
&|_: &str| None,
|
||||
);
|
||||
let transformed = to_value(
|
||||
AZURE_ANTHROPIC_MESSAGES_CONFIG
|
||||
.transform_anthropic_messages_request(request_from(body.clone()))
|
||||
.transform_anthropic_messages_request(request_from(body.clone()), &context)
|
||||
.expect("request transforms"),
|
||||
);
|
||||
assert_eq!(transformed, body);
|
||||
|
|
|
|||
|
|
@ -1,8 +1,14 @@
|
|||
use litellm_http::request::{has_bearer_auth, has_header};
|
||||
use litellm_types::llms::anthropic_messages::{
|
||||
anthropic_request::AnthropicMessagesRequest, anthropic_response::AnthropicMessagesResponse,
|
||||
};
|
||||
|
||||
use crate::base_llm::chat::transformation::Error;
|
||||
use crate::{
|
||||
anthropic::experimental_pass_through::messages::thinking::ThinkingContext,
|
||||
base_llm::chat::transformation::Error,
|
||||
};
|
||||
|
||||
pub type Headers = Vec<(String, String)>;
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum MessagesAuthStrategy {
|
||||
|
|
@ -19,6 +25,14 @@ impl MessagesAuthStrategy {
|
|||
}
|
||||
}
|
||||
|
||||
/// What a provider transformation knows about the call beyond the request body: the
|
||||
/// model's capability flags and the caller's `drop_params` choice.
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||
pub struct MessagesTransformContext {
|
||||
pub thinking: ThinkingContext,
|
||||
pub drop_params: bool,
|
||||
}
|
||||
|
||||
pub trait BaseAnthropicMessagesConfig: Sync {
|
||||
fn get_complete_url(
|
||||
&self,
|
||||
|
|
@ -30,6 +44,7 @@ pub trait BaseAnthropicMessagesConfig: Sync {
|
|||
fn transform_anthropic_messages_request(
|
||||
&self,
|
||||
request: AnthropicMessagesRequest,
|
||||
_context: &MessagesTransformContext,
|
||||
) -> Result<AnthropicMessagesRequest, Error> {
|
||||
Ok(request)
|
||||
}
|
||||
|
|
@ -56,10 +71,39 @@ pub trait BaseAnthropicMessagesConfig: Sync {
|
|||
false
|
||||
}
|
||||
|
||||
/// The forwarded headers with the provider credential applied. A request that already
|
||||
/// carries the provider's auth header (or a bearer the provider accepts) is left alone.
|
||||
fn authenticate(
|
||||
&self,
|
||||
headers: Headers,
|
||||
api_key: Option<&str>,
|
||||
env_lookup: &dyn Fn(&str) -> Option<String>,
|
||||
) -> Result<Headers, Error> {
|
||||
let strategy = self.auth_strategy();
|
||||
if has_header(&headers, strategy.header_name())
|
||||
|| (self.accepts_bearer_auth() && has_bearer_auth(&headers))
|
||||
{
|
||||
return Ok(headers);
|
||||
}
|
||||
let api_key = self.resolve_api_key(api_key, env_lookup)?;
|
||||
let auth_header = match strategy {
|
||||
MessagesAuthStrategy::Bearer => {
|
||||
("authorization".to_string(), format!("Bearer {api_key}"))
|
||||
}
|
||||
MessagesAuthStrategy::Header(name) => (name.to_string(), api_key),
|
||||
};
|
||||
Ok(headers.into_iter().chain([auth_header]).collect())
|
||||
}
|
||||
|
||||
fn default_headers(&self) -> &'static [(&'static str, &'static str)] {
|
||||
&[
|
||||
("anthropic-version", "2023-06-01"),
|
||||
("content-type", "application/json"),
|
||||
]
|
||||
}
|
||||
|
||||
/// Headers the transformed request's features call for, such as `anthropic-beta`.
|
||||
fn request_headers(&self, headers: Headers, _request: &AnthropicMessagesRequest) -> Headers {
|
||||
headers
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ use bytes::Bytes;
|
|||
use litellm_core::messages::{
|
||||
Error,
|
||||
route::{Messages, MessagesCall, MessagesOp, MessagesOpResult, MessagesOutput},
|
||||
types::MessagesShaping,
|
||||
};
|
||||
use litellm_host_python::{InvokeError, RouteHost, from_py, lookup, to_py};
|
||||
use litellm_http::transport::Error as TransportError;
|
||||
|
|
@ -11,6 +12,7 @@ use pyo3::{
|
|||
prelude::*,
|
||||
types::{PyBytes, PyDict},
|
||||
};
|
||||
use serde::Deserialize;
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
use crate::{
|
||||
|
|
@ -18,9 +20,14 @@ use crate::{
|
|||
marshal::{optional_timeout, python_timeout_seconds},
|
||||
};
|
||||
|
||||
const ROUTE_HOST_MODULE: &str = "litellm.rust_bridge.messages.route_host";
|
||||
/// Set on a request rejected before the provider was called, so the Python host maps it to
|
||||
/// the public 400 rather than a connection failure.
|
||||
const REQUEST_ERROR_MARKER: &str = "messages_request_error";
|
||||
|
||||
/// The Anthropic Messages body fields a caller may pass besides `model` and `messages`,
|
||||
/// as `AnthropicMessagesRequestOptionalParams` declares them.
|
||||
const BODY_FIELDS: [&str; 20] = [
|
||||
const BODY_FIELDS: [&str; 22] = [
|
||||
"max_tokens",
|
||||
"metadata",
|
||||
"stop_sequences",
|
||||
|
|
@ -35,14 +42,50 @@ const BODY_FIELDS: [&str; 20] = [
|
|||
"top_p",
|
||||
"mcp_servers",
|
||||
"context_management",
|
||||
"compaction",
|
||||
"container",
|
||||
"output_format",
|
||||
"speed",
|
||||
"output_config",
|
||||
"cache_control",
|
||||
"reasoning_effort",
|
||||
"safeguards",
|
||||
];
|
||||
|
||||
/// One `provider_specific_header` entry: headers scoped to a comma separated provider list.
|
||||
#[derive(Deserialize)]
|
||||
struct ProviderSpecificHeader {
|
||||
#[serde(default)]
|
||||
custom_llm_provider: String,
|
||||
#[serde(default)]
|
||||
extra_headers: Map<String, Value>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
#[serde(untagged)]
|
||||
enum ProviderSpecificHeaders {
|
||||
One(ProviderSpecificHeader),
|
||||
Many(Vec<ProviderSpecificHeader>),
|
||||
}
|
||||
|
||||
impl ProviderSpecificHeaders {
|
||||
fn matching(self, provider: &str) -> impl Iterator<Item = (String, Value)> {
|
||||
let entries = match self {
|
||||
Self::One(entry) => vec![entry],
|
||||
Self::Many(entries) => entries,
|
||||
};
|
||||
entries
|
||||
.into_iter()
|
||||
.filter(move |entry| {
|
||||
entry
|
||||
.custom_llm_provider
|
||||
.split(',')
|
||||
.any(|scoped| scoped.trim() == provider)
|
||||
})
|
||||
.flat_map(|entry| entry.extra_headers)
|
||||
}
|
||||
}
|
||||
|
||||
/// The Python side of the Messages route: projects the prepared arguments and builds the
|
||||
/// public response, chunks and exceptions.
|
||||
pub(super) struct MessagesRouteHost {
|
||||
|
|
@ -84,19 +127,65 @@ impl MessagesRouteHost {
|
|||
.map(|value| python_timeout_seconds(py, value.unbind()))
|
||||
.transpose()?
|
||||
.flatten();
|
||||
let custom_llm_provider = string("custom_llm_provider")?;
|
||||
let shaping = self.shaping(py, &model, custom_llm_provider.as_deref(), arguments)?;
|
||||
Ok(MessagesCall {
|
||||
model,
|
||||
body,
|
||||
api_key: string("api_key")?,
|
||||
api_base: string("api_base")?,
|
||||
custom_llm_provider: string("custom_llm_provider")?,
|
||||
extra_headers: argument("extra_headers")?
|
||||
.map(|value| from_py(&value))
|
||||
.transpose()?,
|
||||
extra_headers: self.merged_headers(py, arguments)?,
|
||||
custom_llm_provider,
|
||||
timeout: optional_timeout(timeout),
|
||||
shaping,
|
||||
})
|
||||
}
|
||||
|
||||
/// Python's handler merges the forwarded `headers`, `extra_headers` and the
|
||||
/// `provider_specific_header` entries scoped to this provider, in that order.
|
||||
fn merged_headers(
|
||||
&self,
|
||||
py: Python<'_>,
|
||||
arguments: &Bound<'_, PyDict>,
|
||||
) -> PyResult<Option<Map<String, Value>>> {
|
||||
let request = self.request.bind(py);
|
||||
let mapping = |name: &str| -> PyResult<Option<Map<String, Value>>> {
|
||||
lookup(arguments, request, name)?
|
||||
.filter(|value| !value.is_none())
|
||||
.map(|value| from_py(&value))
|
||||
.transpose()
|
||||
};
|
||||
let provider = self.provider(py);
|
||||
let scoped = lookup(arguments, request, "provider_specific_header")?
|
||||
.filter(|value| !value.is_none())
|
||||
.map(|value| from_py::<ProviderSpecificHeaders>(&value))
|
||||
.transpose()?
|
||||
.into_iter()
|
||||
.flat_map(|headers| headers.matching(&provider));
|
||||
let merged: Map<String, Value> = mapping("headers")?
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.chain(mapping("extra_headers")?.into_iter().flatten())
|
||||
.chain(scoped)
|
||||
.collect();
|
||||
Ok((!merged.is_empty()).then_some(merged))
|
||||
}
|
||||
|
||||
fn shaping(
|
||||
&self,
|
||||
py: Python<'_>,
|
||||
model: &str,
|
||||
custom_llm_provider: Option<&str>,
|
||||
arguments: &Bound<'_, PyDict>,
|
||||
) -> PyResult<MessagesShaping> {
|
||||
let projected = py.import(ROUTE_HOST_MODULE)?.getattr("shaping")?.call1((
|
||||
model,
|
||||
custom_llm_provider,
|
||||
arguments,
|
||||
))?;
|
||||
from_py(&projected)
|
||||
}
|
||||
|
||||
fn provider(&self, py: Python<'_>) -> String {
|
||||
self.request
|
||||
.bind(py)
|
||||
|
|
@ -112,7 +201,7 @@ impl MessagesRouteHost {
|
|||
return error;
|
||||
}
|
||||
let mapped = py
|
||||
.import("litellm.rust_bridge.messages.route_host")
|
||||
.import(ROUTE_HOST_MODULE)
|
||||
.and_then(|module| module.getattr("map_failure"))
|
||||
.and_then(|map| map.call1((error.value(py), self.request.bind(py), self.provider(py))))
|
||||
.and_then(|mapped| {
|
||||
|
|
@ -148,7 +237,7 @@ impl RouteHost for MessagesRouteHost {
|
|||
fn complete(&mut self, py: Python<'_>, response: MessagesOutput) -> PyResult<Py<PyAny>> {
|
||||
match response {
|
||||
MessagesOutput::Message(message) => py
|
||||
.import("litellm.rust_bridge.messages.route_host")?
|
||||
.import(ROUTE_HOST_MODULE)?
|
||||
.getattr("response")?
|
||||
.call1((to_py(py, message.as_ref())?,))
|
||||
.map(Bound::unbind),
|
||||
|
|
@ -169,6 +258,11 @@ impl RouteHost for MessagesRouteHost {
|
|||
.setattr("headers", Vec::<(String, String)>::new())?;
|
||||
error
|
||||
}
|
||||
Error::InvalidRequest(message) => {
|
||||
let error = PyValueError::new_err(message);
|
||||
error.value(py).setattr(REQUEST_ERROR_MARKER, true)?;
|
||||
error
|
||||
}
|
||||
other => messages_error_to_pyerr(other),
|
||||
};
|
||||
Ok(self.map_failure(py, native))
|
||||
|
|
|
|||
|
|
@ -15,14 +15,52 @@ pub enum MessageContent {
|
|||
Blocks(Vec<ContentBlock>),
|
||||
}
|
||||
|
||||
/// One content block of a message or system prompt. The fields the route reads are typed;
|
||||
/// everything else rides in `extra` untouched.
|
||||
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
|
||||
pub struct ContentBlock {
|
||||
#[serde(rename = "type", default, skip_serializing_if = "Option::is_none")]
|
||||
pub block_type: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub text: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub thinking: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub signature: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub data: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub id: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub name: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub tool_use_id: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub input: Option<Value>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub content: Option<Value>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub provider_specific_fields: Option<Value>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_control: Option<CacheControl>,
|
||||
#[serde(flatten)]
|
||||
pub extra: Map<String, Value>,
|
||||
}
|
||||
|
||||
impl ContentBlock {
|
||||
pub fn text(text: impl Into<String>) -> Self {
|
||||
Self {
|
||||
block_type: Some("text".to_string()),
|
||||
text: Some(text.into()),
|
||||
..Self::default()
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_type(&self, block_type: &str) -> bool {
|
||||
self.block_type.as_deref() == Some(block_type)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
|
||||
pub struct CacheControl {
|
||||
#[serde(rename = "type", skip_serializing_if = "Option::is_none")]
|
||||
|
|
@ -85,6 +123,26 @@ pub struct AnthropicMessagesRequest {
|
|||
pub speed: Option<String>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub inference_geo: Option<String>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub reasoning_effort: Option<String>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub compaction: Option<Value>,
|
||||
#[serde(flatten)]
|
||||
pub extra: Map<String, Value>,
|
||||
}
|
||||
|
||||
impl AnthropicMessage {
|
||||
pub fn blocks(&self) -> &[ContentBlock] {
|
||||
match &self.content {
|
||||
MessageContent::Blocks(blocks) => blocks,
|
||||
MessageContent::Text(_) => &[],
|
||||
}
|
||||
}
|
||||
|
||||
pub fn with_blocks(self, blocks: Vec<ContentBlock>) -> Self {
|
||||
Self {
|
||||
content: MessageContent::Blocks(blocks),
|
||||
..self
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,12 +1,51 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping
|
||||
from typing import cast # noqa: TID251 # narrows the normalized native payload to the public TypedDict
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Final, cast # noqa: TID251 # narrows the normalized native payload to the public TypedDict
|
||||
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.core_helpers import normalize_drop_params
|
||||
from litellm.llms.anthropic.experimental_pass_through.utils import is_reasoning_auto_summary_enabled
|
||||
from litellm.rust_bridge import failures
|
||||
from litellm.rust_bridge.messages.entrypoints import LiteLLMMessagesRequest
|
||||
from litellm.types.llms.anthropic_messages.anthropic_response import AnthropicMessagesResponse
|
||||
|
||||
_EFFORT_TIERS: Final = ("minimal", "low", "medium", "high", "xhigh", "max")
|
||||
_DROP_PATHS: Final = TypeAdapter(list[object])
|
||||
_CAPABILITY_FLAGS: Final = (
|
||||
"supports_reasoning",
|
||||
"supports_adaptive_thinking",
|
||||
"thinking_always_on",
|
||||
"supports_legacy_thinking",
|
||||
"supports_output_config",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ModelCapabilities:
|
||||
supports_reasoning: bool
|
||||
supports_adaptive_thinking: bool
|
||||
thinking_always_on: bool
|
||||
supports_legacy_thinking: bool
|
||||
supports_output_config: bool
|
||||
supports_sampling_params: bool
|
||||
supports_speed: bool
|
||||
effort_tiers: Mapping[str, bool]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MessagesShaping:
|
||||
"""What the native Messages route needs from the Python side to shape a request the way
|
||||
the Python handler does: the model's cost-map flags and the caller's LiteLLM settings."""
|
||||
|
||||
capabilities: ModelCapabilities
|
||||
drop_params: bool
|
||||
reasoning_auto_summary: bool
|
||||
additional_drop_params: Sequence[str]
|
||||
|
||||
|
||||
def response(value: Mapping[str, object]) -> AnthropicMessagesResponse:
|
||||
return cast( # cast-ok: AnthropicMessagesResponse is a TypedDict over the normalized native payload
|
||||
|
|
@ -20,4 +59,63 @@ def arguments(request: LiteLLMMessagesRequest) -> Mapping[str, object]:
|
|||
|
||||
|
||||
def map_failure(error: Exception, request: LiteLLMMessagesRequest, request_provider: str) -> Exception:
|
||||
if getattr(error, "messages_request_error", False):
|
||||
return litellm.BadRequestError(
|
||||
message=str(error),
|
||||
model=request.model.removeprefix(f"{request_provider}/"),
|
||||
llm_provider=request_provider,
|
||||
)
|
||||
return failures.map_native_failure(error, request.model, request_provider, arguments(request), request.api_base)
|
||||
|
||||
|
||||
def _resolved_provider(model: str, custom_llm_provider: str | None) -> tuple[str, str]:
|
||||
try:
|
||||
resolved_model, provider, _, _ = litellm.get_llm_provider(model=model, custom_llm_provider=custom_llm_provider)
|
||||
except Exception: # noqa: BLE001 # an unroutable model still shapes as a bare Anthropic id
|
||||
return model, custom_llm_provider or "anthropic"
|
||||
return resolved_model, provider
|
||||
|
||||
|
||||
def model_capabilities(model: str, custom_llm_provider: str | None) -> ModelCapabilities:
|
||||
"""The flags Python's Anthropic transforms read from the cost map, resolved under the
|
||||
caller's provider exactly as `AnthropicModelInfo._supports_model_capability` does."""
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
resolved_model, provider = _resolved_provider(model, custom_llm_provider)
|
||||
flags: Final = {
|
||||
flag: AnthropicModelInfo._supports_model_capability(model, flag, provider) # pyright: ignore[reportPrivateUsage] # same probes the Python transform runs; forking them would drift
|
||||
for flag in _CAPABILITY_FLAGS
|
||||
}
|
||||
return ModelCapabilities(
|
||||
supports_sampling_params=AnthropicModelInfo._supports_sampling_params(resolved_model), # pyright: ignore[reportPrivateUsage] # same gate the handler applies
|
||||
supports_speed=AnthropicConfig._model_supports_speed_param(resolved_model, provider), # pyright: ignore[reportPrivateUsage] # same gate the handler applies
|
||||
effort_tiers={
|
||||
tier: AnthropicConfig._supports_effort_level(model, tier, provider) # pyright: ignore[reportPrivateUsage] # same probe the Python transform runs
|
||||
for tier in _EFFORT_TIERS
|
||||
},
|
||||
**flags,
|
||||
)
|
||||
|
||||
|
||||
def _drop_params(kwargs: Mapping[str, object]) -> bool:
|
||||
return bool(litellm.drop_params) or normalize_drop_params(kwargs.get("drop_params")) is True
|
||||
|
||||
|
||||
def _additional_drop_params(kwargs: Mapping[str, object]) -> tuple[str, ...]:
|
||||
try:
|
||||
configured: Final = _DROP_PATHS.validate_python(kwargs.get("additional_drop_params"))
|
||||
except ValidationError:
|
||||
return ()
|
||||
return tuple(path for path in configured if isinstance(path, str))
|
||||
|
||||
|
||||
def shaping(model: str, custom_llm_provider: str | None, kwargs: Mapping[str, object]) -> dict[str, object]:
|
||||
return asdict( # mutable-ok: the native side depythonizes a plain dict
|
||||
MessagesShaping(
|
||||
capabilities=model_capabilities(model, custom_llm_provider),
|
||||
drop_params=_drop_params(kwargs),
|
||||
reasoning_auto_summary=is_reasoning_auto_summary_enabled(),
|
||||
additional_drop_params=_additional_drop_params(kwargs),
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -5,3 +5,5 @@ Test what each side of the bridge does, not the rollout policy that picks a side
|
|||
Call each path directly with an explicit decision instead. The Python path is the implementation the dispatcher falls back to, e.g. `litellm.ocr.main.ocr`. The Rust path is the native binding, e.g. `NATIVE_OCR.load()` from `litellm/rust_bridge/ocr/entrypoints.py`, called with the request, args and kwargs that dispatch would hand it. When the native side reads a policy-derived setting such as `settings.secret_manager().native`, pin that field in the test instead of deriving it from the catalog. `ocr/test_secrets.py` shows the pattern
|
||||
|
||||
Rollout policy itself, meaning which rule matches and what `LITELLM_RUST` changes, belongs in `test_catalog.py`, `test_configuration.py` and `test_dispatch.py`, tested against rules the test builds rather than the shipped `catalog.RULES`
|
||||
|
||||
Before adding a test here, ask whether it checks something Rust cannot. A `route_host.py` module is the Python half of a native route: it projects Python-only state (the cost map, `litellm.*` settings, request kwargs) into the plain values the Rust side consumes, and maps native failures back onto public exceptions. Those projections are what belongs here, because a wrong key or an ignored provider prefix ships the wrong value to Rust and no Rust test sees it. `messages/test_route_host.py` shows the shape. Behavior that lives in Rust (a request transform given its inputs, header assembly, stream relay) is tested in the crate, and the route end to end is tested against a recording server in `tests/test_litellm_rust/`. A test that only re-checks a Python helper the route host happens to call is a duplicate of that helper's own test and should not be added
|
||||
|
|
|
|||
111
tests/test_litellm/rust_bridge/messages/test_route_host.py
Normal file
111
tests/test_litellm/rust_bridge/messages/test_route_host.py
Normal file
|
|
@ -0,0 +1,111 @@
|
|||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.rust_bridge.messages import route_host
|
||||
|
||||
pytestmark = pytest.mark.usefixtures("local_model_cost_map")
|
||||
|
||||
|
||||
def _flag_model(monkeypatch: pytest.MonkeyPatch, name: str, **flags: bool) -> None:
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
name,
|
||||
{
|
||||
"litellm_provider": "anthropic",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 0,
|
||||
"output_cost_per_token": 0,
|
||||
**flags,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def test_capabilities_come_from_the_model_map_under_the_callers_provider(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
_flag_model(
|
||||
monkeypatch,
|
||||
"claude-test-adaptive",
|
||||
supports_reasoning=True,
|
||||
supports_adaptive_thinking=True,
|
||||
supports_output_config=True,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
supports_sampling_params=False,
|
||||
)
|
||||
|
||||
capabilities: Final = route_host.model_capabilities("anthropic/claude-test-adaptive", None)
|
||||
|
||||
assert capabilities.supports_adaptive_thinking
|
||||
assert capabilities.supports_output_config
|
||||
assert not capabilities.supports_legacy_thinking
|
||||
assert not capabilities.supports_sampling_params
|
||||
assert capabilities.effort_tiers["xhigh"]
|
||||
assert not capabilities.effort_tiers["max"]
|
||||
|
||||
|
||||
def test_unmapped_model_keeps_sampling_params_and_no_reasoning_features() -> None:
|
||||
capabilities: Final = route_host.model_capabilities("anthropic/not-a-real-model", None)
|
||||
|
||||
assert capabilities.supports_sampling_params
|
||||
assert not capabilities.supports_reasoning
|
||||
assert not capabilities.supports_adaptive_thinking
|
||||
assert not any(capabilities.effort_tiers.values())
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("global_flag", "kwargs", "expected"),
|
||||
[
|
||||
(False, {}, False),
|
||||
(True, {}, True),
|
||||
(False, {"drop_params": "true"}, True),
|
||||
(False, {"drop_params": "nonsense"}, False),
|
||||
(False, {"drop_params": False}, False),
|
||||
],
|
||||
)
|
||||
def test_drop_params_merges_the_global_flag_with_the_request(
|
||||
monkeypatch: pytest.MonkeyPatch, global_flag: bool, kwargs: dict[str, object], expected: bool
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "drop_params", global_flag)
|
||||
|
||||
assert route_host.shaping("anthropic/not-a-real-model", None, kwargs)["drop_params"] is expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("configured", "expected"),
|
||||
[
|
||||
(["tools[*].input_examples", 3, "metadata.user_id"], ("tools[*].input_examples", "metadata.user_id")),
|
||||
("tools", ()),
|
||||
(None, ()),
|
||||
],
|
||||
)
|
||||
def test_additional_drop_params_keep_only_string_paths(configured: object, expected: tuple[str, ...]) -> None:
|
||||
shaping: Final = route_host.shaping("anthropic/not-a-real-model", None, {"additional_drop_params": configured})
|
||||
|
||||
assert shaping["additional_drop_params"] == expected
|
||||
|
||||
|
||||
def test_native_request_rejections_map_to_the_public_400() -> None:
|
||||
from types import MappingProxyType
|
||||
|
||||
from litellm.rust_bridge.messages.entrypoints import LiteLLMMessagesRequest
|
||||
|
||||
request: Final = LiteLLMMessagesRequest(
|
||||
model="anthropic/claude-sonnet-5",
|
||||
messages=(),
|
||||
max_tokens=8,
|
||||
stream=None,
|
||||
api_key=None,
|
||||
api_base=None,
|
||||
custom_llm_provider=None,
|
||||
kwargs=MappingProxyType({}),
|
||||
)
|
||||
rejected: Final = ValueError("claude-sonnet-5 does not support top_k=5")
|
||||
rejected.messages_request_error = True # pyright: ignore[reportAttributeAccessIssue] # marker the native host sets
|
||||
|
||||
mapped: Final = route_host.map_failure(rejected, request, "anthropic")
|
||||
|
||||
assert isinstance(mapped, litellm.BadRequestError)
|
||||
assert mapped.status_code == 400
|
||||
assert "does not support top_k=5" in mapped.message
|
||||
assert mapped.model == "claude-sonnet-5"
|
||||
assert not isinstance(route_host.map_failure(ValueError("plain"), request, "anthropic"), litellm.BadRequestError)
|
||||
204
tests/test_litellm_rust/messages/test_request_shaping.py
Normal file
204
tests/test_litellm_rust/messages/test_request_shaping.py
Normal file
|
|
@ -0,0 +1,204 @@
|
|||
"""The native Messages route shapes the wire request the way the Python handler does.
|
||||
|
||||
Model capability expectations come from model_prices_and_context_window.json (Claude Sonnet 5 is an
|
||||
adaptive-thinking model without sampling params; Claude Haiku 4.5 is a legacy-thinking model), read at
|
||||
2026-09-24; the cost map is LiteLLM's own file.
|
||||
"""
|
||||
|
||||
from collections.abc import Iterator
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.rust_bridge import catalog
|
||||
from litellm.rust_bridge.catalog import Route, RouteRule
|
||||
from litellm.rust_bridge.configuration import Rollout
|
||||
from tests.test_litellm_rust.support.isolation import rebound
|
||||
from tests.test_litellm_rust.support.recording_server import RecordingServer, ResponseSpec
|
||||
from tests.test_litellm_rust.support.requests import MESSAGES, MESSAGES_RESPONSE
|
||||
|
||||
pytestmark = pytest.mark.requires_rust_extension
|
||||
|
||||
ADAPTIVE_MODEL: Final = "anthropic/claude-sonnet-5"
|
||||
LEGACY_THINKING_MODEL: Final = "anthropic/claude-haiku-4-5"
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def opt_messages_into_rust() -> Iterator[None]:
|
||||
with rebound(catalog, "RULES", (RouteRule(Route.MESSAGES, Rollout.RUST_OPT_IN), *catalog.RULES)):
|
||||
yield
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def messages_server(recording_server: RecordingServer) -> RecordingServer:
|
||||
recording_server.default_response = ResponseSpec(body=MESSAGES_RESPONSE)
|
||||
return recording_server
|
||||
|
||||
|
||||
def arguments(server: RecordingServer, **kwargs: object) -> dict[str, object]:
|
||||
return {
|
||||
"model": ADAPTIVE_MODEL,
|
||||
"messages": [dict(message) for message in MESSAGES],
|
||||
"max_tokens": 8192,
|
||||
"api_key": "test-key",
|
||||
"api_base": server.base_url,
|
||||
**kwargs,
|
||||
}
|
||||
|
||||
|
||||
def sent(server: RecordingServer) -> tuple[dict[str, object], dict[str, str]]:
|
||||
assert len(server.requests) == 1
|
||||
request: Final = server.requests[0]
|
||||
assert not request.headers.get("user-agent", "").startswith("python-httpx")
|
||||
assert isinstance(request.body, dict)
|
||||
return request.body, request.headers
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_reasoning_effort_becomes_adaptive_thinking_and_effort_on_the_wire(
|
||||
messages_server: RecordingServer,
|
||||
) -> None:
|
||||
await litellm.anthropic.messages.acreate(**arguments(messages_server, reasoning_effort="high"))
|
||||
|
||||
body, _ = sent(messages_server)
|
||||
assert "reasoning_effort" not in body
|
||||
assert body["thinking"] == {"type": "adaptive", "display": "summarized"}
|
||||
assert body["output_config"] == {"effort": "high"}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_claude_code_adaptive_payload_is_downgraded_to_a_capped_budget_for_a_legacy_model(
|
||||
messages_server: RecordingServer,
|
||||
) -> None:
|
||||
await litellm.anthropic.messages.acreate(
|
||||
**arguments(
|
||||
messages_server,
|
||||
model=LEGACY_THINKING_MODEL,
|
||||
max_tokens=3000,
|
||||
thinking={"type": "adaptive"},
|
||||
output_config={"effort": "high"},
|
||||
temperature=0,
|
||||
)
|
||||
)
|
||||
|
||||
body, _ = sent(messages_server)
|
||||
assert body["thinking"] == {"type": "enabled", "budget_tokens": 2999}
|
||||
assert "output_config" not in body
|
||||
assert "temperature" not in body
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_removed_sampling_params_are_dropped_under_drop_params(messages_server: RecordingServer) -> None:
|
||||
await litellm.anthropic.messages.acreate(
|
||||
**arguments(messages_server, temperature=0.2, top_p=0.9, top_k=5, drop_params=True)
|
||||
)
|
||||
|
||||
body, _ = sent(messages_server)
|
||||
assert not {"temperature", "top_p", "top_k"} & body.keys()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_removed_sampling_params_are_rejected_without_drop_params(messages_server: RecordingServer) -> None:
|
||||
messages_server.expected_requests = 0
|
||||
|
||||
with pytest.raises(litellm.BadRequestError, match="does not support top_k=5"):
|
||||
await litellm.anthropic.messages.acreate(**arguments(messages_server, top_k=5))
|
||||
|
||||
assert messages_server.requests == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_replayed_history_is_sanitized_before_it_reaches_the_provider(
|
||||
messages_server: RecordingServer,
|
||||
) -> None:
|
||||
history: Final = [
|
||||
{"role": "user", "content": "run it"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "text", "text": ""},
|
||||
{
|
||||
"type": "tool_use",
|
||||
"id": "functions.Bash:0",
|
||||
"name": "Bash",
|
||||
"input": {},
|
||||
"provider_specific_fields": {"x": 1},
|
||||
},
|
||||
],
|
||||
},
|
||||
{"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions.Bash:0", "content": "ok"}]},
|
||||
]
|
||||
|
||||
await litellm.anthropic.messages.acreate(**arguments(messages_server, messages=history))
|
||||
|
||||
body, _ = sent(messages_server)
|
||||
assert body["messages"] == [
|
||||
{"role": "user", "content": "run it"},
|
||||
{"role": "assistant", "content": [{"type": "tool_use", "id": "functions_Bash_0", "name": "Bash", "input": {}}]},
|
||||
{"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions_Bash_0", "content": "ok"}]},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_feature_betas_merge_into_the_forwarded_beta_header(messages_server: RecordingServer) -> None:
|
||||
await litellm.anthropic.messages.acreate(
|
||||
**arguments(
|
||||
messages_server,
|
||||
output_format={"type": "json_schema", "schema": {"type": "object"}},
|
||||
extra_headers={"anthropic-beta": "web-search-2025-03-05"},
|
||||
)
|
||||
)
|
||||
|
||||
_, headers = sent(messages_server)
|
||||
assert headers["anthropic-beta"] == "structured-outputs-2025-11-13,web-search-2025-03-05"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_oauth_token_authenticates_as_a_bearer_with_the_oauth_beta(messages_server: RecordingServer) -> None:
|
||||
await litellm.anthropic.messages.acreate(**arguments(messages_server, api_key="sk-ant-oat01-token"))
|
||||
|
||||
_, headers = sent(messages_server)
|
||||
assert "x-api-key" not in headers
|
||||
assert headers["authorization"] == "Bearer sk-ant-oat01-token"
|
||||
assert headers["anthropic-beta"] == "oauth-2025-04-20"
|
||||
assert headers["anthropic-dangerous-direct-browser-access"] == "true"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_metadata_is_reduced_to_the_fields_anthropic_accepts(messages_server: RecordingServer) -> None:
|
||||
await litellm.anthropic.messages.acreate(
|
||||
**arguments(messages_server, metadata={"user_id": "u-1", "trace_id": "internal"})
|
||||
)
|
||||
|
||||
body, _ = sent(messages_server)
|
||||
assert body["metadata"] == {"user_id": "u-1"}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_additional_drop_params_remove_nested_fields_from_the_wire(messages_server: RecordingServer) -> None:
|
||||
tools: Final = [{"name": "lookup", "input_schema": {"type": "object"}, "input_examples": [{"q": "x"}]}]
|
||||
|
||||
await litellm.anthropic.messages.acreate(
|
||||
**arguments(messages_server, tools=tools, additional_drop_params=["tools[*].input_examples"])
|
||||
)
|
||||
|
||||
body, _ = sent(messages_server)
|
||||
assert body["tools"] == [{"name": "lookup", "input_schema": {"type": "object"}}]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_specific_headers_scoped_to_anthropic_reach_the_wire(messages_server: RecordingServer) -> None:
|
||||
await litellm.anthropic.messages.acreate(
|
||||
**arguments(
|
||||
messages_server,
|
||||
provider_specific_header=[
|
||||
{"custom_llm_provider": "anthropic, azure_ai", "extra_headers": {"x-scoped": "yes"}},
|
||||
{"custom_llm_provider": "openai", "extra_headers": {"x-other": "no"}},
|
||||
],
|
||||
)
|
||||
)
|
||||
|
||||
_, headers = sent(messages_server)
|
||||
assert headers["x-scoped"] == "yes"
|
||||
assert "x-other" not in headers
|
||||
Loading…
Add table
Reference in a new issue