refactor(rust): add shared llms wire type derives (#43730)

This commit is contained in:
devin-ai-integration[bot] 2026-09-29 09:19:33 -07:00 • committed by GitHub
parent 684a1edd44
commit 66db132627
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
143 changed files with 1404 additions and 1321 deletions

View file

@ -3619,7 +3619,6 @@ dependencies = [
"litellm-auth",
"litellm-host",
"litellm-host-python",
"litellm-types",
"proptest",
"pyo3",
"rstest",
@ -3658,9 +3657,9 @@ dependencies = [
"litellm-host-native",
"litellm-http",
"litellm-llms",
"litellm-llms-types",
"litellm-secrets",
"litellm-tracing",
"litellm-types",
"mime_guess",
"moka",
"rand 0.8.7",
@ -3688,13 +3687,12 @@ name = "litellm-core-utils"
version = "0.1.0"
dependencies = [
"fancy-regex 0.19.2",
"litellm-llms-types",
"litellm-tracing",
"litellm-types",
"rstest",
"serde",
"serde_json",
"serde_path_to_error",
"serde_with",
"strum",
"thiserror 2.0.19",
"url",
@ -3824,9 +3822,9 @@ dependencies = [
"litellm-host-http",
"litellm-http",
"litellm-llms",
"litellm-llms-types",
"litellm-router",
"litellm-secrets",
"litellm-types",
"rstest",
"serde",
"serde_json",
@ -3999,9 +3997,9 @@ dependencies = [
"litellm-framing",
"litellm-host",
"litellm-http",
"litellm-llms-types",
"litellm-python-compat",
"litellm-secrets",
"litellm-types",
"reqwest 0.12.28",
"rstest",
"serde",
@ -4015,13 +4013,26 @@ dependencies = [
"url",
]
[[package]]
name = "litellm-llms-types"
version = "0.1.0"
dependencies = [
"macro_rules_attribute",
"rstest",
"schemars 1.2.2",
"serde",
"serde_json",
"serde_with",
"strum",
]
[[package]]
name = "litellm-model-catalog"
version = "0.1.0"
dependencies = [
"indexmap 2.14.0",
"jsonschema",
"litellm-types",
"litellm-llms-types",
"rstest",
"schemars 1.2.2",
"serde",
@ -4059,12 +4070,12 @@ dependencies = [
"litellm-host-python",
"litellm-http",
"litellm-llms",
"litellm-llms-types",
"litellm-secrets",
"litellm-secrets-aws",
"litellm-secrets-types",
"litellm-token-counter",
"litellm-tracing",
"litellm-types",
"pyo3",
"pyo3-async-runtimes",
"qdrant-client",
@ -4354,17 +4365,6 @@ dependencies = [
"tracing-subscriber",
]
[[package]]
name = "litellm-types"
version = "0.1.0"
dependencies = [
"rstest",
"schemars 1.2.2",
"serde",
"serde_json",
"strum",
]
[[package]]
name = "litemap"
version = "0.8.2"

View file

@ -39,7 +39,7 @@ litellm-secrets-azure = { path = "crates/secrets-azure" }
litellm-secrets-cyberark = { path = "crates/secrets-cyberark" }
litellm-http = { path = "crates/http" }
litellm-llms = { path = "crates/llms" }
litellm-types = { path = "crates/types" }
litellm-llms-types = { path = "crates/llms-types" }
litellm-core-utils = { path = "crates/core-utils" }
litellm-db = { path = "crates/db" }
litellm-db-testing = { path = "crates/db-testing" }
@ -74,6 +74,7 @@ proptest = "1.7.0"
pyo3 = "0.29.2"
pyo3-async-runtimes = { version = "0.29.0", features = ["tokio-runtime"] }
rand = "0.8"
macro_rules_attribute = "0.2.3"
schemars = "1"
reqwest = { version = "0.12", default-features = false, features = ["json", "multipart", "rustls-tls", "http2", "stream"] }
qdrant-client = { version = "1.19.0", default-features = false }

View file

@ -7,6 +7,7 @@
- The enum only shrinks: when Rust owns a subsystem, delete its group rather than adding a Rust path beside it
- Calling a user's own callback directly is permanent Python surface and gets its own type outside `LegacyPython`
- `PublicCall` is the caller's call as `Logging` sees it: the positional arguments, the keyword view as the call rewrites it (setup, deployment hook, preflight) and the bound request object backing omitted keywords; shared bridge composition hands it to `LegacyLogging`; routes use the neutral call boundary
- `LoggingOperation` selects legacy logging entrypoints and response handling. It belongs here rather than in shared inference data contracts
- `setup` reuses a `Logging` passed as `litellm_logging_obj` (the proxy and Router) and otherwise builds one through `function_setup`; which callbacks run is `Logging`'s decision, never this crate's
- Callbacks receive the caller's own objects and may mutate them; this crate alone carries that obligation
- Retain complete boundary arguments, opaque values, aliases, omitted/default distinctions and deliberate copies; preserve the deployment-hook kwargs view

View file

@ -6,7 +6,6 @@ license.workspace = true
repository.workspace = true
[dependencies]
litellm-types.workspace = true
litellm-host.workspace = true
litellm-host-python.workspace = true

View file

@ -2,8 +2,8 @@
//! raises is answered with the same `Logging` calls, in the same order, as the Python
//! `@client` path makes them.
use crate::LoggingOperation;
use litellm_host_python::PythonOwned;
use litellm_types::Operation;
use litellm_host::{
interceptors::{RawResponse, RequestContext, WireRequest},
@ -45,7 +45,7 @@ struct LoggedRequest {
}
pub struct LegacyLogging {
operation: Operation,
operation: LoggingOperation,
call: PublicCall,
logger: Option<PythonLogger>,
start: Py<PyAny>,
@ -68,7 +68,12 @@ fn is_cancellation(py: Python<'_>, error: &PyErr) -> bool {
}
impl LegacyLogging {
pub fn new(py: Python<'_>, operation: Operation, call: PublicCall, asynchronous: bool) -> Self {
pub fn new(
py: Python<'_>,
operation: LoggingOperation,
call: PublicCall,
asynchronous: bool,
) -> Self {
Self {
operation,
call,
@ -87,32 +92,34 @@ impl LegacyLogging {
fn call_type(&self) -> &'static str {
match (self.operation, self.asynchronous) {
(Operation::Completion, false) => "completion",
(Operation::Completion, true) => "acompletion",
(Operation::Responses, false) => "responses",
(Operation::Responses, true) => "aresponses",
(Operation::Messages, _) => "anthropic_messages",
(Operation::Ocr, false) => "ocr",
(Operation::Ocr, true) => "aocr",
(LoggingOperation::Completion, false) => "completion",
(LoggingOperation::Completion, true) => "acompletion",
(LoggingOperation::Responses, false) => "responses",
(LoggingOperation::Responses, true) => "aresponses",
(LoggingOperation::Messages, _) => "anthropic_messages",
(LoggingOperation::Ocr, false) => "ocr",
(LoggingOperation::Ocr, true) => "aocr",
}
}
fn input_description(&self) -> &'static str {
match self.operation {
Operation::Completion => "Chat completions",
Operation::Responses => "Responses",
Operation::Messages => "Messages",
Operation::Ocr => "OCR document processing",
LoggingOperation::Completion => "Chat completions",
LoggingOperation::Responses => "Responses",
LoggingOperation::Messages => "Messages",
LoggingOperation::Ocr => "OCR document processing",
}
}
fn stream_billing(&self) -> Option<PassThroughStream> {
match self.operation {
Operation::Messages => Some(PassThroughStream {
LoggingOperation::Messages => Some(PassThroughStream {
url_route: "/v1/messages",
endpoint_type: "anthropic",
}),
Operation::Completion | Operation::Responses | Operation::Ocr => None,
LoggingOperation::Completion | LoggingOperation::Responses | LoggingOperation::Ocr => {
None
}
}
}
@ -643,16 +650,16 @@ kwargs = {'logger': logger, 'document': document}
}
#[rstest]
#[case::sync_completion(litellm_types::Operation::Completion, false, "completion")]
#[case::async_completion(litellm_types::Operation::Completion, true, "acompletion")]
#[case::sync_responses(litellm_types::Operation::Responses, false, "responses")]
#[case::async_responses(litellm_types::Operation::Responses, true, "aresponses")]
#[case::sync_messages(litellm_types::Operation::Messages, false, "anthropic_messages")]
#[case::async_messages(litellm_types::Operation::Messages, true, "anthropic_messages")]
#[case::sync_ocr(litellm_types::Operation::Ocr, false, "ocr")]
#[case::async_ocr(litellm_types::Operation::Ocr, true, "aocr")]
#[case::sync_completion(crate::LoggingOperation::Completion, false, "completion")]
#[case::async_completion(crate::LoggingOperation::Completion, true, "acompletion")]
#[case::sync_responses(crate::LoggingOperation::Responses, false, "responses")]
#[case::async_responses(crate::LoggingOperation::Responses, true, "aresponses")]
#[case::sync_messages(crate::LoggingOperation::Messages, false, "anthropic_messages")]
#[case::async_messages(crate::LoggingOperation::Messages, true, "anthropic_messages")]
#[case::sync_ocr(crate::LoggingOperation::Ocr, false, "ocr")]
#[case::async_ocr(crate::LoggingOperation::Ocr, true, "aocr")]
fn operation_selects_the_legacy_setup_and_deployment_hook_contract(
#[case] operation: litellm_types::Operation,
#[case] operation: crate::LoggingOperation,
#[case] asynchronous: bool,
#[case] expected: &str,
) {
@ -1088,12 +1095,12 @@ check = lambda: None
}
#[rstest]
#[case::completion(litellm_types::Operation::Completion, "Chat completions")]
#[case::responses(litellm_types::Operation::Responses, "Responses")]
#[case::messages(litellm_types::Operation::Messages, "Messages")]
#[case::ocr(litellm_types::Operation::Ocr, "OCR document processing")]
#[case::completion(crate::LoggingOperation::Completion, "Chat completions")]
#[case::responses(crate::LoggingOperation::Responses, "Responses")]
#[case::messages(crate::LoggingOperation::Messages, "Messages")]
#[case::ocr(crate::LoggingOperation::Ocr, "OCR document processing")]
fn prepared_arguments_replace_the_legacy_view_without_losing_callback_aliases(
#[case] operation: litellm_types::Operation,
#[case] operation: crate::LoggingOperation,
#[case] description: &str,
) {
Python::initialize();
@ -1763,7 +1770,7 @@ assert logger.calls[1][1] is response
Python::attach(|py| {
let locals = namespace(py, c"first = b'first'\nlast = b'last'\nresponse = None");
let mut logging = LegacyLogging {
operation: litellm_types::Operation::Messages,
operation: crate::LoggingOperation::Messages,
..logged(py, &locals, true)
};
logging

View file

@ -20,5 +20,13 @@ pub(crate) use callbacks::{LegacyCallbacks, is_internal_call};
pub(crate) use logger::{DeploymentHooks, PythonLogger, finalize, setup};
pub use mapping::{CallBoundary, CallbackMapping, Dispatch, callback_mappings};
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum LoggingOperation {
Completion,
Responses,
Messages,
Ocr,
}
#[cfg(test)]
mod test_support;

View file

@ -189,5 +189,5 @@ pub(crate) fn legacy_call(
.map(|kwargs| kwargs.cast_into::<PyDict>().unwrap())
.unwrap_or_else(|| PyDict::new(py));
let call = PublicCall::capture(&request, &PyTuple::empty(py), &kwargs).unwrap();
LegacyLogging::new(py, litellm_types::Operation::Ocr, call, asynchronous)
LegacyLogging::new(py, crate::LoggingOperation::Ocr, call, asynchronous)
}

View file

@ -8,11 +8,10 @@ repository.workspace = true
[dependencies]
fancy-regex.workspace = true
litellm-tracing.workspace = true
litellm-types.workspace = true
litellm-llms-types.workspace = true
serde.workspace = true
serde_json.workspace = true
serde_path_to_error = "0.1"
serde_with.workspace = true
strum.workspace = true
thiserror.workspace = true
url.workspace = true

View file

@ -2,7 +2,7 @@
use std::time::{SystemTime, UNIX_EPOCH};
use litellm_types::utils::{ChatCompletionsUsage, PromptTokensDetails};
use litellm_llms_types::formats::chat_completions::{ChatCompletionsUsage, PromptTokensDetails};
/// OpenAI finish reasons, mirroring Python's `_FINISH_REASON_MAP` for the
/// reasons the providers on this route can emit. Python warns and falls back to

View file

@ -1,4 +1,4 @@
use litellm_types::utils::{ProviderSpecificHeader, ProviderSpecificHeaders};
use litellm_llms_types::headers::{ProviderSpecificHeader, ProviderSpecificHeaders};
use serde_json::{Map, Value};
pub fn get_provider_specific_headers(

View file

@ -10,7 +10,7 @@
//! `_bedrock_converse_messages_pt` for the text-only surface this route
//! accepts; anything richer is declined upstream by the capability gate.
use litellm_types::llms::openai::{ChatMessage, ChatMessageContent};
use litellm_llms_types::formats::chat_completions::{ChatMessage, ChatMessageContent};
use strum::IntoStaticStr;
pub const EMPTY_TEXT_PLACEHOLDER: &str =

View file

@ -1,12 +1,3 @@
use serde::{
Deserializer,
de::{Error, Visitor},
};
use serde_with::DeserializeAs;
pub struct LaxI64;
pub struct FiniteF64;
pub fn parse_str_bool(value: &str) -> Option<bool> {
let token = value.trim_matches(|character: char| {
character.is_whitespace() || matches!(character, '\u{1c}'..='\u{1f}')
@ -22,129 +13,12 @@ pub fn parse_redis_bool(value: &str) -> bool {
value == "1" || value.eq_ignore_ascii_case("true") || value.eq_ignore_ascii_case("yes")
}
impl<'de> DeserializeAs<'de, i64> for LaxI64 {
fn deserialize_as<D: Deserializer<'de>>(deserializer: D) -> Result<i64, D::Error> {
deserializer.deserialize_any(Self)
}
}
impl<'de> Visitor<'de> for LaxI64 {
type Value = i64;
fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
formatter.write_str("an integer in the i64 range")
}
fn visit_i64<E: Error>(self, value: i64) -> Result<i64, E> {
Ok(value)
}
fn visit_u64<E: Error>(self, value: u64) -> Result<i64, E> {
i64::try_from(value).map_err(E::custom)
}
fn visit_f64<E: Error>(self, value: f64) -> Result<i64, E> {
integral_float(value).ok_or_else(|| E::custom("expected an integer in the i64 range"))
}
fn visit_str<E: Error>(self, value: &str) -> Result<i64, E> {
integer_string(value.trim())
.ok_or_else(|| E::custom("expected an integer in the i64 range"))
}
fn visit_bool<E: Error>(self, value: bool) -> Result<i64, E> {
Ok(i64::from(value))
}
}
impl<'de> DeserializeAs<'de, f64> for FiniteF64 {
fn deserialize_as<D: Deserializer<'de>>(deserializer: D) -> Result<f64, D::Error> {
deserializer.deserialize_any(Self)
}
}
impl<'de> Visitor<'de> for FiniteF64 {
type Value = f64;
fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
formatter.write_str("a finite number")
}
fn visit_i64<E: Error>(self, value: i64) -> Result<f64, E> {
Ok(value as f64)
}
fn visit_u64<E: Error>(self, value: u64) -> Result<f64, E> {
Ok(value as f64)
}
fn visit_f64<E: Error>(self, value: f64) -> Result<f64, E> {
value
.is_finite()
.then_some(value)
.ok_or_else(|| E::custom("expected a finite number"))
}
fn visit_str<E: Error>(self, value: &str) -> Result<f64, E> {
self.visit_f64(value.trim().parse::<f64>().map_err(E::custom)?)
}
fn visit_bool<E: Error>(self, value: bool) -> Result<f64, E> {
Ok(f64::from(value))
}
}
fn integer_string(value: &str) -> Option<i64> {
let integer = match value.split_once('.') {
Some((integer, fraction)) => {
if fraction.is_empty() || !fraction.bytes().all(|byte| byte == b'0') {
return None;
}
integer
}
None => value,
};
if integer.starts_with('_') || integer.ends_with('_') || integer.contains("__") {
return None;
}
let digits = integer.strip_prefix(['+', '-']).unwrap_or(integer);
if digits.is_empty()
|| digits.starts_with('_')
|| !digits
.bytes()
.all(|byte| byte.is_ascii_digit() || byte == b'_')
{
return None;
}
integer.replace('_', "").parse().ok()
}
fn integral_float(value: f64) -> Option<i64> {
(value.is_finite()
&& value.fract() == 0.0
&& value >= i64::MIN as f64
&& value < -(i64::MIN as f64))
.then_some(value as i64)
}
#[cfg(test)]
mod tests {
use rstest::rstest;
use serde::{Deserialize, Serialize};
use serde_json::json;
use serde_with::serde_as;
use super::*;
#[serde_as]
#[derive(Debug, Deserialize, Serialize, PartialEq)]
struct Numbers {
#[serde_as(deserialize_as = "Option<Vec<LaxI64>>")]
integers: Option<Vec<i64>>,
#[serde_as(deserialize_as = "Option<FiniteF64>")]
float: Option<f64>,
}
#[rstest]
#[case::trimmed_true(" True ", Some(true))]
#[case::control_whitespace_true("\u{1c}TRUE\u{1f}", Some(true))]
@ -160,73 +34,4 @@ mod tests {
) {
assert_eq!(parse_str_bool(input), expected, "{input:?}");
}
#[test]
fn adapters_compose_and_serialize_as_numbers() {
let numbers: Numbers = serde_json::from_value(json!({
"integers": ["9007199254740993.0", "1_000", " +2.000 ", 3.0, true],
"float": " 1.5 "
}))
.unwrap();
assert_eq!(
serde_json::to_value(numbers).unwrap(),
json!({
"integers": [9_007_199_254_740_993_i64, 1000, 2, 3, 1], "float": 1.5
})
);
for input in [json!({}), json!({"integers": null, "float": null})] {
assert_eq!(
serde_json::from_value::<Numbers>(input).unwrap(),
Numbers {
integers: None,
float: None,
}
);
}
}
#[test]
fn integer_bounds_and_invalid_values_are_checked() {
for input in [
json!(i64::MIN),
json!(i64::MAX),
json!(i64::MAX.to_string()),
] {
assert!(serde_json::from_value::<Numbers>(json!({"integers": [input]})).is_ok());
}
for input in [
json!(u64::MAX),
json!(9_223_372_036_854_775_808_u64),
json!(9_223_372_036_854_775_808.0),
json!("-9223372036854775809"),
json!("1.0000000000000001"),
json!("1e3"),
json!("2."),
json!(".0"),
json!("_2"),
json!("2__0"),
json!(2.5),
json!(null),
json!({}),
] {
assert!(serde_json::from_value::<Numbers>(json!({"integers": [input]})).is_err());
}
}
#[test]
fn floats_reject_nonfinite_and_invalid_values() {
for input in [
json!("NaN"),
json!("inf"),
json!("-inf"),
json!("1e999"),
json!([]),
] {
assert!(serde_json::from_value::<Numbers>(json!({"float": input})).is_err());
}
for (input, expected) in [(json!(2), 2.0), (json!(2.5), 2.5), (json!(true), 1.0)] {
let numbers: Numbers = serde_json::from_value(json!({"float": input})).unwrap();
assert_eq!(numbers.float, Some(expected));
}
}
}

View file

@ -10,11 +10,11 @@ Responses WebSocket sessions remain separate from the HTTP call driver because a
## Crate layering
For Messages, Responses, Chat Completions, OCR, and other API formats, `core/src/<format>/` owns orchestration. Shared API data contracts belong in `litellm-types`, adapter contracts and shared transformation machinery in `llms/src/base_llm/<format>/`, and provider policy in `llms/src/<provider>/<format>/`. A repeated format directory name does not imply interchangeable responsibilities. Select concrete adapters here, then invoke their contracts instead of applying one provider's policy to every call. Route types describe call envelopes and execution state, not duplicate public payload schemas
For Messages, Responses, Chat Completions, OCR, and other API formats, `core/src/<format>/` owns orchestration. Shared API data contracts belong in `litellm-llms-types`, adapter contracts and shared transformation machinery in `llms/src/base_llm/<format>/`, and provider policy in `llms/src/<provider>/<format>/`. A repeated format directory name does not imply interchangeable responsibilities. Select concrete adapters here, then invoke their contracts instead of applying one provider's policy to every call. Route types describe call envelopes and execution state, not duplicate public payload schemas
Each crate mirrors one top-level Python package, so a Rust path reads as its Python path with the crate name in place of the package directory. Dependencies only point down:
Crates separate API data, transformations, transport, and orchestration. Python package names identify counterparts, not ownership. Dependencies only point down:
- `litellm-types` mirrors `litellm/types/`: pure serde data, no I/O
- `litellm-llms-types` owns shared inference API contracts, grouped by format: pure serde data and shape validation, no I/O
- `litellm-core-utils` mirrors `litellm/litellm_core_utils/`: pure helpers (provider resolution, prompt factory, call arguments, settings lookup and layer merge), no network I/O
- `litellm-http` is Rust-only and route-neutral: settings resolution, the pooled `reqwest` clients, TLS, proxies, the SSRF-safe media fetcher, request and header helpers, and transport errors. Python's `litellm/llms/custom_httpx/` is split by responsibility instead of mirrored: its transport half lives here, its OCR handler in `litellm-llms`
- `litellm-llms` mirrors `litellm/llms/`: `base_llm/<api>/transformation.rs`, `<provider>/<api>/transformation.rs`, and `base_llm/ocr/handler.rs` (the OCR request handler)

View file

@ -11,7 +11,7 @@ litellm-cache-response.workspace = true
litellm-framing.workspace = true
tokio-util = { version = "0.7", features = ["codec"] }
litellm-secrets.workspace = true
litellm-types.workspace = true
litellm-llms-types.workspace = true
litellm-core-utils.workspace = true
litellm-host.workspace = true
bytes.workspace = true

View file

@ -1,5 +1,4 @@
use litellm_host::lifecycle::ExecutionEvent;
use litellm_host::observation::ObservationSender;
use litellm_host::{lifecycle::ExecutionEvent, observation::ObservationSender};
use std::time::Duration;
use litellm_auth::AuthServices;
@ -9,7 +8,7 @@ use litellm_llms::base_llm::{
auth::{Authenticated, resolve_auth},
chat::transformation::ProviderChatResponseData,
};
use litellm_types::utils::ChatCompletionsResponse;
use litellm_llms_types::formats::chat_completions::ChatCompletionsResponse;
use serde_json::Value;
use super::Error;

View file

@ -5,7 +5,7 @@ pub use crate::error::RouteError as Error;
mod common_utils;
pub(crate) mod handler;
mod prepare;
use litellm_types::utils::ChatCompletionsResponse;
use litellm_llms_types::formats::chat_completions::ChatCompletionsResponse;
use prepare::{prepare_provider_request, resolve_request};
use crate::chat_completions::types::ChatCompletionsRequest;

View file

@ -2,8 +2,8 @@ use litellm_auth::SecretValue;
use litellm_core_utils::settings::Lookup;
use litellm_http::request::with_default_headers;
use litellm_llms::base_llm::{auth::ValidatedEnvironment, chat::transformation::BaseConfig};
use litellm_llms_types::formats::chat_completions::ChatMessage;
use litellm_secrets::source::Secrets;
use litellm_types::llms::openai::ChatMessage;
use serde_json::Value;
use super::{

View file

@ -5,7 +5,7 @@ use litellm_host::{
call::{CallOutput, HostedMachine, hosted_call},
protocol::Protocol,
};
use litellm_types::utils::ChatCompletionsResponse;
use litellm_llms_types::formats::chat_completions::ChatCompletionsResponse;
use super::{
ChatCompletionsRoute, Error,

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use litellm_auth::SecretValue;
use litellm_llms::base_llm::{auth::ValidatedEnvironment, chat::transformation::BaseConfig};
use litellm_types::llms::openai::ChatMessage;
use litellm_llms_types::formats::chat_completions::ChatMessage;
use serde_json::{Map, Value};
/// A `/chat/completions` call as it crosses into the core.

View file

@ -1,4 +1,4 @@
This directory owns provider-independent Messages call orchestration: the entrypoint, call envelopes, provider selection, credential resolution, transport coordination, hooks, and stream lifecycle. Shared API data contracts belong in `litellm-types::messages`, adapter contracts and execution inputs in `llms/src/base_llm/messages`, and provider implementations in `llms/src/<provider>/messages`
This directory owns provider-independent Messages call orchestration: the entrypoint, call envelopes, provider selection, credential resolution, transport coordination, hooks, and stream lifecycle. Shared API data contracts belong in `litellm-llms-types::formats::messages`, adapter contracts and execution inputs in `llms/src/base_llm/messages`, and provider implementations in `llms/src/<provider>/messages`
Select concrete provider adapters and invoke their contracts. Delegate authentication policy, beta selection, payload rewriting, and response interpretation to those adapters. Keep provider policy out of request preparation and transport handlers. Calling a concrete provider helper for every provider is still a policy dependency

View file

@ -3,7 +3,7 @@ pub(super) use litellm_http::request::truncate_error_body;
use litellm_llms::{
anthropic::messages::transformation::ANTHROPIC_MESSAGES_CONFIG,
azure_ai::messages::transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG,
base_llm::messages::transformation::BaseAnthropicMessagesConfig,
base_llm::messages::transformation::BaseMessagesConfig,
bedrock::messages::invoke_transformations::anthropic_claude3_transformation::BEDROCK_ANTHROPIC_MESSAGES_CONFIG,
};
use serde_json::{Map, Value};
@ -30,7 +30,7 @@ impl MessagesProvider {
.into()
}
pub(crate) fn config(self) -> &'static dyn BaseAnthropicMessagesConfig {
pub(crate) fn config(self) -> &'static dyn BaseMessagesConfig {
match self {
Self::Anthropic => &ANTHROPIC_MESSAGES_CONFIG,
Self::AzureAi => &AZURE_ANTHROPIC_MESSAGES_CONFIG,

View file

@ -1,5 +1,4 @@
use litellm_host::lifecycle::ExecutionEvent;
use litellm_host::observation::ObservationSender;
use litellm_host::{lifecycle::ExecutionEvent, observation::ObservationSender};
use std::time::Duration;
use bytes::Bytes;
@ -11,15 +10,16 @@ use litellm_llms::base_llm::{
auth::{Authenticated, resolve_auth},
messages::{
streaming::{ByteStream, StreamDecoder, encode_anthropic_sse},
transformation::BaseAnthropicMessagesConfig,
transformation::BaseMessagesConfig,
},
};
use litellm_llms_types::formats::messages::MessagesResponse;
use litellm_tracing::ByteChunk;
use litellm_types::llms::anthropic_messages::anthropic_response::AnthropicMessagesResponse;
use serde_json::Value;
use super::{
Error, MessagesResponse, common_utils::truncate_error_body, prepare::ProviderMessagesRequest,
Error, MessagesCallResponse, common_utils::truncate_error_body,
prepare::ProviderMessagesRequest,
};
use crate::{constants::MESSAGES_TIMEOUT_SECS, outbound::outbound_request};
@ -31,7 +31,7 @@ pub(super) async fn execute(
cache_options: Option<litellm_cache_response::CacheOptions>,
interceptors: &impl Interceptors<Error>,
observers: Option<&ObservationSender>,
) -> Result<MessagesResponse, Error> {
) -> Result<MessagesCallResponse, Error> {
let ProviderMessagesRequest {
provider,
url,
@ -110,7 +110,7 @@ pub(super) async fn execute(
.await
.map_err(Error::post_call)?;
decode_response(config, &body.model, &text)
.map(|message| MessagesResponse::Complete(Box::new(message)))
.map(|message| MessagesCallResponse::Complete(Box::new(message)))
},
)
.await
@ -158,10 +158,10 @@ async fn provider_error(response: reqwest::Response) -> Error {
}
fn decode_response(
config: &dyn BaseAnthropicMessagesConfig,
config: &dyn BaseMessagesConfig,
model: &str,
text: &str,
) -> Result<AnthropicMessagesResponse, Error> {
) -> Result<MessagesResponse, Error> {
let response = serde_json::from_str(text).map_err(|err| {
Error::InvalidResponse(litellm_llms::ErrorDetail::invalid(
"messages response JSON",
@ -177,7 +177,7 @@ fn streaming_response(
response: reqwest::Response,
decoder: Option<StreamDecoder>,
provider: &'static str,
) -> MessagesResponse {
) -> MessagesCallResponse {
let headers = response
.headers()
.iter()
@ -194,7 +194,7 @@ fn streaming_response(
.boxed(),
Some(decode) => decoded_chunks(response, decode, provider),
};
MessagesResponse::Stream {
MessagesCallResponse::Stream {
head: super::route::MessagesStreamHead { headers },
chunks,
}
@ -268,7 +268,7 @@ mod tests {
.send()
.await
.unwrap();
let MessagesResponse::Stream { mut chunks, .. } =
let MessagesCallResponse::Stream { mut chunks, .. } =
streaming_response(response, Some(anthropic_sse_event_stream), "test")
else {
panic!("a streaming response returns chunks");

View file

@ -10,7 +10,7 @@ use litellm_secrets::source::SecretSource;
use std::sync::Arc;
pub use crate::error::RouteError as Error;
pub use types::{MessagesCall, MessagesResponse, MessagesShaping, messages_body};
pub use types::{MessagesCall, MessagesCallResponse, MessagesShaping, messages_body};
#[derive(Clone)]
pub struct MessagesRoute {
@ -103,7 +103,7 @@ impl MessagesRoute {
call: MessagesCall,
interceptors: &impl litellm_host::interceptors::Interceptors<Error>,
options: impl Into<crate::CallOptions>,
) -> Result<MessagesResponse, Error> {
) -> Result<MessagesCallResponse, Error> {
let crate::CallOptions {
cache: cache_options,
observers,
@ -129,7 +129,7 @@ impl MessagesRoute {
cache_options: Option<litellm_cache_response::CacheOptions>,
interceptors: &impl litellm_host::interceptors::Interceptors<Error>,
observers: Option<&ObservationSender>,
) -> Result<MessagesResponse, Error> {
) -> Result<MessagesCallResponse, Error> {
crate::diagnostic::call(async {
self.run_provider(call, cache_options, interceptors, observers)
.await
@ -143,10 +143,10 @@ impl MessagesRoute {
cache_options: Option<litellm_cache_response::CacheOptions>,
interceptors: &impl litellm_host::interceptors::Interceptors<Error>,
observers: Option<&ObservationSender>,
) -> Result<MessagesResponse, Error> {
) -> Result<MessagesCallResponse, Error> {
let request = prepare::prepare(call, self.secrets.as_ref()).await?;
crate::diagnostic::provider(&request.body.model, request.provider.as_str());
let execute: futures_util::future::BoxFuture<'_, Result<MessagesResponse, Error>> =
let execute: futures_util::future::BoxFuture<'_, Result<MessagesCallResponse, Error>> =
Box::pin(handler::execute(
&self.http,
&self.auth,

View file

@ -9,8 +9,8 @@ use litellm_http::request::with_default_headers;
use litellm_llms::base_llm::{
auth::ValidatedEnvironment, messages::context::MessagesTransformContext,
};
use litellm_llms_types::formats::messages::MessagesRequest;
use litellm_secrets::source::SecretSource;
use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest;
use super::{
Error, MessagesCall,
@ -27,7 +27,7 @@ struct ResolvedProvider {
pub(super) struct ProviderMessagesRequest {
pub(super) provider: MessagesProvider,
pub(super) url: String,
pub(super) body: AnthropicMessagesRequest,
pub(super) body: MessagesRequest,
pub(super) environment: ValidatedEnvironment,
pub(super) timeout: Option<Duration>,
/// The caller's own credential, reported to the host beside the wire request.
@ -79,7 +79,7 @@ fn prepare_provider_request(
let env_lookup = |key: &str| secrets.get(key);
let sanitized = config.shape_request(
AnthropicMessagesRequest { model, ..body },
MessagesRequest { model, ..body },
shaping.reasoning_auto_summary,
)?;
let trimmed = without_additional_drop_params(sanitized, &shaping.additional_drop_params)?;
@ -124,9 +124,9 @@ fn prepare_provider_request(
}
fn without_additional_drop_params(
request: AnthropicMessagesRequest,
request: MessagesRequest,
paths: &[String],
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
if paths.is_empty() {
return Ok(request);
}
@ -134,7 +134,7 @@ fn without_additional_drop_params(
let trimmed = paths
.iter()
.fold(params, |params, path| delete_nested_value(params, path));
Ok(AnthropicMessagesRequest {
Ok(MessagesRequest {
params: serde_json::from_value(trimmed).map_err(invalid_request)?,
..request
})
@ -143,7 +143,7 @@ fn without_additional_drop_params(
#[cfg(test)]
mod tests {
use litellm_llms::base_llm::auth::resolve_auth;
use litellm_types::utils::ProviderSpecificHeaders;
use litellm_llms_types::headers::ProviderSpecificHeaders;
use rstest::{fixture, rstest};
use serde_json::{Map, Value, json};
@ -155,7 +155,7 @@ mod tests {
MessagesShaping::default()
}
fn body(value: Value) -> AnthropicMessagesRequest {
fn body(value: Value) -> MessagesRequest {
serde_json::from_value(value).unwrap()
}

View file

@ -5,11 +5,11 @@ use litellm_host::{
call::{HostedCompletion, HostedMachine, hosted_call},
protocol::Protocol,
};
use litellm_types::llms::anthropic_messages::anthropic_response::AnthropicMessagesResponse;
use litellm_llms_types::formats::messages::MessagesResponse;
use super::{Error, MessagesCall};
pub type MessagesOutput = HostedCompletion<Box<AnthropicMessagesResponse>>;
pub type MessagesOutput = HostedCompletion<Box<MessagesResponse>>;
/// The upstream response as the caller sees it at stream hand-off, before any chunk.
pub struct MessagesStreamHead {
@ -19,7 +19,7 @@ pub struct MessagesStreamHead {
pub struct Messages;
impl Protocol for Messages {
type Response = Box<AnthropicMessagesResponse>;
type Response = Box<MessagesResponse>;
type Error = Error;
type Request = MessagesCall;
type HostCall = Infallible;

View file

@ -2,12 +2,10 @@ use std::time::Duration;
use bytes::Bytes;
use litellm_host::call::CallOutput;
use litellm_llms::base_llm::messages::context::MessagesModelCapabilities as AnthropicModelCapabilities;
use litellm_types::{
llms::anthropic_messages::{
anthropic_request::AnthropicMessagesRequest, anthropic_response::AnthropicMessagesResponse,
},
utils::ProviderSpecificHeaders,
use litellm_llms::base_llm::messages::context::MessagesModelCapabilities;
use litellm_llms_types::{
formats::messages::{MessagesRequest, MessagesResponse},
headers::ProviderSpecificHeaders,
};
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
@ -15,7 +13,7 @@ use serde_json::{Map, Value};
use super::Error;
pub struct MessagesCall {
pub body: AnthropicMessagesRequest,
pub body: MessagesRequest,
pub api_key: Option<String>,
pub api_base: Option<String>,
pub custom_llm_provider: Option<String>,
@ -25,7 +23,7 @@ pub struct MessagesCall {
pub shaping: MessagesShaping,
}
pub fn messages_body(body: Map<String, Value>) -> Result<AnthropicMessagesRequest, Error> {
pub fn messages_body(body: Map<String, Value>) -> Result<MessagesRequest, Error> {
serde_json::from_value(Value::Object(body)).map_err(invalid_request)
}
@ -33,13 +31,13 @@ pub(super) fn invalid_request(err: serde_json::Error) -> Error {
Error::InvalidRequest(format!("invalid Anthropic messages request: {err}").into())
}
pub type MessagesResponse =
CallOutput<Box<AnthropicMessagesResponse>, super::route::MessagesStreamHead, Bytes, Error>;
pub type MessagesCallResponse =
CallOutput<Box<MessagesResponse>, super::route::MessagesStreamHead, Bytes, Error>;
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
pub struct MessagesShaping {
#[serde(default)]
pub capabilities: AnthropicModelCapabilities,
pub capabilities: MessagesModelCapabilities,
#[serde(default)]
pub drop_params: bool,
#[serde(default)]
@ -76,9 +74,9 @@ mod tests {
#[case::partial_capabilities(
json!({"capabilities": {"supports_reasoning": true}}),
MessagesShaping {
capabilities: AnthropicModelCapabilities {
capabilities: MessagesModelCapabilities {
supports_reasoning: true,
..AnthropicModelCapabilities::default()
..MessagesModelCapabilities::default()
},
..MessagesShaping::default()
},
@ -100,7 +98,7 @@ mod tests {
"additional_drop_params": ["metadata.user_id", "thinking"]
}),
MessagesShaping {
capabilities: AnthropicModelCapabilities {
capabilities: MessagesModelCapabilities {
supports_reasoning: true,
supports_adaptive_thinking: true,
thinking_always_on: false,

View file

@ -2,9 +2,8 @@ use litellm_host::observation::ObservationSender;
use std::sync::Arc;
use litellm_host::interceptors::Interceptors;
use litellm_llms::base_llm::ocr::{
error::Error, handler::OcrClient, transformation::LiteLLMOcrResponse,
};
use litellm_llms::base_llm::ocr::{error::Error, handler::OcrClient};
use litellm_llms_types::formats::ocr::LiteLLMOcrResponse;
use super::{
handler::perform_ocr_request,

View file

@ -1,10 +1,8 @@
use std::{collections::BTreeMap as Map, io::Read, path::Path};
use base64::{Engine, engine::general_purpose::STANDARD};
use litellm_llms::base_llm::ocr::{
error::Error,
transformation::{OCR_INLINE_MAX_BYTES, OcrDocument},
};
use litellm_llms::base_llm::ocr::{error::Error, transformation::OCR_INLINE_MAX_BYTES};
use litellm_llms_types::formats::ocr::OcrDocument;
use crate::ocr::types::OcrDocumentInput;

View file

@ -1,12 +1,12 @@
use futures_util::future::BoxFuture;
use litellm_host::interceptors::{Interceptors, RawResponse, RequestContext, WireRequest};
use litellm_host::lifecycle::ExecutionEvent;
use litellm_host::observation::ObservationSender;
use litellm_host::{lifecycle::ExecutionEvent, observation::ObservationSender};
use litellm_llms::base_llm::ocr::{
error::Error,
handler::{CallHooks, OcrClient},
transformation::{LiteLLMOcrResponse, PreparedOcrRequest},
transformation::PreparedOcrRequest,
};
use litellm_llms_types::formats::ocr::LiteLLMOcrResponse;
use serde_json::Value;
use super::{arguments::is_secret_param, prepare::prepare_request, provider_config::OcrConfigKind};

View file

@ -85,12 +85,13 @@ mod tests {
base_llm::ocr::{
error::Error,
handler::{CallHooks, OcrClient},
transformation::{BaseOcrConfig, OcrResponseFormat},
transformation::BaseOcrConfig,
},
cohere::ocr::transformation::CohereParseConfig,
mistral::ocr::transformation::MistralOcrConfig,
vertex_ai::ocr::transformation::VertexAiOcrConfig,
};
use litellm_llms_types::formats::ocr::OcrResponseFormat;
use serde_json::{Value, json};
use super::*;

View file

@ -14,8 +14,7 @@ use litellm_llms::{
error::Error,
handler::{self, CallHooks, OcrClient},
transformation::{
BaseOcrConfig, LiteLLMOcrResponse, OcrCredentialInputs, OcrDocument, OcrResponseFormat,
PreparedOcrRequest, ResolvedOcrCredentials,
BaseOcrConfig, OcrCredentialInputs, PreparedOcrRequest, ResolvedOcrCredentials,
},
},
cohere::ocr::transformation::CohereParseConfig,
@ -25,6 +24,7 @@ use litellm_llms::{
deepseek_transformation::VertexAIDeepSeekOCRConfig, transformation::VertexAiOcrConfig,
},
};
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument, OcrResponseFormat};
macro_rules! with_config {
($kind:expr, $config:ident => $body:expr) => {

View file

@ -6,7 +6,8 @@ use litellm_host::{
protocol::Protocol,
protocol::Reply,
};
use litellm_llms::base_llm::ocr::{error::Error, transformation::LiteLLMOcrResponse};
use litellm_llms::base_llm::ocr::error::Error;
use litellm_llms_types::formats::ocr::LiteLLMOcrResponse;
use crate::ocr::types::{LiteLLMOcrRequest, OcrDocumentInput};

View file

@ -5,10 +5,9 @@ use litellm_auth::{InputSource, SecretValue, TokenProviderHandle};
use litellm_core_utils::call_arguments::CallArguments;
use litellm_llms::base_llm::ocr::{
error::Error,
transformation::{
OcrCredentialInputs, OcrDocument, OcrResponseFormat, OcrTransportConfig, response_format,
},
transformation::{OcrCredentialInputs, OcrTransportConfig, response_format},
};
use litellm_llms_types::formats::ocr::{OcrDocument, OcrResponseFormat};
use serde_json::{Map, Value};
use super::provider_config::{OcrConfigKind, resolve_provider_config};
@ -222,7 +221,7 @@ mod tests {
use super::*;
fn document() -> OcrDocument {
OcrDocument::try_from(
serde_json::from_value(
json!({"type":"document_url","document_url":"data:application/pdf;base64,YWJj"}),
)
.unwrap()

View file

@ -1,10 +1,8 @@
use std::{collections::BTreeMap, time::Duration};
use litellm_auth::{InputSource, SecretValue};
use litellm_llms::base_llm::ocr::{
error::Error,
transformation::{OcrDocument, decode_request_value},
};
use litellm_llms::base_llm::ocr::{error::Error, transformation::decode_request_value};
use litellm_llms_types::formats::ocr::OcrDocument;
use serde::Deserialize;
use serde_json::{Map, Value};

View file

@ -5,7 +5,7 @@ use litellm_host::{
call::{HostedMachine, hosted_call},
protocol::Protocol,
};
use litellm_types::responses::main::ResponsesApiResponse;
use litellm_llms_types::formats::responses::ResponsesApiResponse;
use super::{
Error, ResponsesRoute,

View file

@ -5,7 +5,7 @@ use litellm_host::call::CallOutput;
use litellm_llms::base_llm::{
auth::ValidatedEnvironment, responses::transformation::BaseResponsesApiConfig,
};
use litellm_types::responses::main::ResponsesApiResponse;
use litellm_llms_types::formats::responses::ResponsesApiResponse;
use serde_json::{Map, Value};
use super::Error;

View file

@ -2,7 +2,7 @@ use std::{collections::HashMap, sync::Arc, time::Duration};
use futures_util::{SinkExt, StreamExt};
use litellm_http::websocket::{UpstreamWebSocket, connect_upstream};
use litellm_types::responses::streaming_websocket::ResponsesWsEventType;
use litellm_llms_types::formats::responses::streaming_websocket::ResponsesWsEventType;
use tokio::sync::Mutex;
use tokio_tungstenite::tungstenite::{
Message,

View file

@ -411,7 +411,7 @@ async fn responses_refetches_instead_of_deserializing_another_api_response(
#[case] poisoned: Value,
) {
use litellm_core::responses::route::Responses;
use litellm_types::responses::main::ResponsesApiResponse;
use litellm_llms_types::formats::responses::ResponsesApiResponse;
let cache: Arc<dyn ResponseCacheService> = Arc::new(InvalidEntryCache(
ResponseCache::new(Arc::new(InMemoryCache::default())),
@ -461,7 +461,7 @@ async fn messages_cache_identity_includes_provider_native_parameters(
#[case] changed: Value,
) {
use litellm_core::messages::route::Messages;
use litellm_types::llms::anthropic_messages::anthropic_response::AnthropicMessagesResponse;
use litellm_llms_types::formats::messages::MessagesResponse;
let calls = AtomicUsize::new(0);
for (value, expected_call) in [(original.clone(), 0), (changed, 1), (original, 0)] {
@ -487,7 +487,7 @@ async fn messages_cache_identity_includes_provider_native_parameters(
None,
|| async {
let call = calls.fetch_add(1, Ordering::SeqCst);
Ok(Box::new(serde_json::from_value::<AnthropicMessagesResponse>(json!({
Ok(Box::new(serde_json::from_value::<MessagesResponse>(json!({
"id":call.to_string(), "type":"message", "role":"assistant", "model":"test",
"content":[{"type":"text","text":format!("answer {call}")}],
"stop_reason":"end_turn", "stop_sequence":null
@ -843,7 +843,7 @@ async fn responses_cache_only_reuses_completed_responses(
#[case] expected_calls: usize,
) {
use litellm_core::responses::route::Responses;
use litellm_types::responses::main::ResponsesApiResponse;
use litellm_llms_types::formats::responses::ResponsesApiResponse;
let calls = AtomicUsize::new(0);
for _ in 0..2 {

View file

@ -7,7 +7,7 @@ use std::time::Duration;
use litellm_core::chat_completions::{Error, types::ChatCompletionsRequest};
use litellm_http::transport::Error as TransportError;
use litellm_types::utils::ChatCompletionsResponse;
use litellm_llms_types::formats::chat_completions::ChatCompletionsResponse;
use rstest::{fixture, rstest};
use serde_json::{Map, Value, json};
use wiremock::ResponseTemplate;

View file

@ -8,10 +8,8 @@ use litellm_core::messages::{
route::{Messages, MessagesMachine, MessagesOutput},
};
use litellm_http::{HttpSettings, Resolution};
use litellm_llms_types::formats::messages::{MessagesRequest, MessagesResponse};
use litellm_secrets::source::SecretSource;
use litellm_types::llms::anthropic_messages::{
anthropic_request::AnthropicMessagesRequest, anthropic_response::AnthropicMessagesResponse,
};
use rstest::fixture;
use serde_json::{Map, Value, json};
use wiremock::ResponseTemplate;
@ -35,7 +33,7 @@ fn object(value: Value) -> Map<String, Value> {
map
}
fn body(value: Value) -> AnthropicMessagesRequest {
fn body(value: Value) -> MessagesRequest {
serde_json::from_value(value).unwrap()
}
@ -116,7 +114,7 @@ async fn run(call: MessagesCall) -> Result<MessagesOutput, Error> {
run_with(Arc::new(RecordingSecrets::empty()), call).await
}
async fn run_message(call: MessagesCall) -> AnthropicMessagesResponse {
async fn run_message(call: MessagesCall) -> MessagesResponse {
match run(call).await.expect("messages call succeeds") {
MessagesOutput::Complete(message) => *message,
MessagesOutput::StreamEnded | MessagesOutput::Detached => {

View file

@ -1,6 +1,8 @@
use litellm_llms::base_llm::messages::context::{MessagesModelCapabilities, SupportedEffortTiers};
use litellm_types::llms::anthropic::{AnthropicBeta, BetaSet};
use litellm_types::utils::{ProviderSpecificHeader, ProviderSpecificHeaders};
use litellm_llms_types::{
headers::{ProviderSpecificHeader, ProviderSpecificHeaders},
providers::anthropic::{AnthropicBeta, BetaSet},
};
use rstest::rstest;
use super::*;

View file

@ -1,4 +1,4 @@
use litellm_core::messages::{MessagesResponse, messages_body};
use litellm_core::messages::{MessagesCallResponse, messages_body};
use litellm_host::{
interceptors::{ExecutionFacts, ResultSource},
lifecycle::ExecutionEvent,
@ -33,7 +33,7 @@ async fn calls_defer_execution_until_polled(
let request = host.request().unwrap();
let observer: Option<litellm_host::observation::ObservationSender> =
with_observer.then(|| host.events.0.sender.clone());
let future: BoxFuture<'_, Result<MessagesResponse, Error>> = if with_hooks {
let future: BoxFuture<'_, Result<MessagesCallResponse, Error>> = if with_hooks {
Box::pin(route.execute(request, &host, observer))
} else {
Box::pin(route.execute(request, &(), observer))
@ -43,7 +43,7 @@ async fn calls_defer_execution_until_polled(
assert!(host.events.0.lock().unwrap().is_empty());
assert!(received(&upstream).await.is_empty());
let MessagesResponse::Complete(response) = future.await.unwrap() else {
let MessagesCallResponse::Complete(response) = future.await.unwrap() else {
panic!("expected a completed message");
};
assert_eq!(
@ -289,7 +289,7 @@ async fn the_facade_sends_through_the_injected_http_pool_configuration(call: Mes
.await
.expect("messages request succeeds");
let MessagesResponse::Complete(message) = response else {
let MessagesCallResponse::Complete(message) = response else {
panic!("a non-streaming request returns a message");
};
assert_eq!(message.id, "msg_1");
@ -380,7 +380,8 @@ async fn builder_preserves_dependencies_and_optional_cache(
api_base: Some(upstream.uri()),
..super::call()
};
let MessagesResponse::Complete(response) = route.execute(request, &(), None).await.unwrap()
let MessagesCallResponse::Complete(response) =
route.execute(request, &(), None).await.unwrap()
else {
panic!("expected a completed message");
};

View file

@ -6,7 +6,7 @@ use std::{
use bytes::Bytes;
use futures_util::{StreamExt, TryStreamExt};
use litellm_core::messages::{
MessagesResponse,
MessagesCallResponse,
route::{Messages, MessagesStreamHead},
};
use litellm_tracing::{Logger, Metadata, Record, Sink};
@ -353,7 +353,7 @@ async fn the_sdk_returns_stream_headers_and_every_sse_byte(
.await
.unwrap();
let MessagesResponse::Stream { head, chunks } = response else {
let MessagesCallResponse::Stream { head, chunks } = response else {
panic!("a streaming request returns a stream");
};
for (name, value) in UPSTREAM_HEADERS {
@ -407,7 +407,7 @@ async fn dropping_the_sdk_stream_closes_the_unfinished_upstream(
.expect("messages() returns before the upstream finishes")
.unwrap();
let MessagesResponse::Stream { mut chunks, .. } = response else {
let MessagesCallResponse::Stream { mut chunks, .. } = response else {
panic!("a streaming request returns a stream");
};
if read_chunk {
@ -442,7 +442,7 @@ async fn the_sdk_yields_a_body_error_once_after_delivered_chunks(call: MessagesC
.await
.unwrap();
let MessagesResponse::Stream { mut chunks, .. } = response else {
let MessagesCallResponse::Stream { mut chunks, .. } = response else {
panic!("a streaming request returns a stream");
};
assert_eq!(

View file

@ -9,11 +9,8 @@ use litellm_host::{
interceptors::{RequestContext, WireRequest},
lifecycle::CallEvent,
};
use litellm_llms::base_llm::ocr::{
error::Error,
settings::OcrSettings,
transformation::{LiteLLMOcrResponse, OcrDocument},
};
use litellm_llms::base_llm::ocr::{error::Error, settings::OcrSettings};
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument};
use serde_json::{Map, Value, json};
use std::sync::Mutex;
use wiremock::{MockServer, ResponseTemplate};

View file

@ -19,7 +19,7 @@ litellm-http.workspace = true
litellm-llms.workspace = true
litellm-router.workspace = true
litellm-secrets.workspace = true
litellm-types.workspace = true
litellm-llms-types.workspace = true
serde.workspace = true
serde_json.workspace = true
thiserror.workspace = true

View file

@ -12,7 +12,7 @@ use axum::{
};
use litellm_core::messages::{MessagesCall, messages_body, route::Messages};
use litellm_host_http::Sse;
use litellm_types::utils::{ProviderSpecificHeader, ProviderSpecificHeaders};
use litellm_llms_types::headers::{ProviderSpecificHeader, ProviderSpecificHeaders};
use serde_json::{Map, Value};
use crate::{Deployment, Error, Gateway, JsonObject, RequestId, request};

View file

@ -4,7 +4,8 @@ use std::sync::Arc;
use axum::{Json, extract::State, http::HeaderMap, response::IntoResponse};
use litellm_auth::SecretValue;
use litellm_core::ocr::types::{LiteLLMOcrRequest, OcrConnectionInputs, OcrDocumentInput};
use litellm_llms::base_llm::ocr::transformation::OcrDocument;
use litellm_llms::base_llm::ocr::transformation::decode_request_value;
use litellm_llms_types::formats::ocr::OcrDocument;
use serde_json::Value;
use crate::{
@ -42,7 +43,11 @@ async fn handle(
file_name: upload.file_name,
mime_type: upload.mime_type,
},
None => OcrDocument::try_from(body.get("document").cloned().unwrap_or_default())?.into(),
None => decode_request_value::<OcrDocument>(
body.get("document").cloned().unwrap_or_default(),
"document",
)?
.into(),
};
let format = body
.get("req_format")

View file

@ -2,7 +2,8 @@ mod support;
use axum::{body::Body, http::Request};
use litellm_gateway_inference::Error;
use litellm_llms::base_llm::ocr::{error::Error as OcrError, transformation::OcrDocument};
use litellm_llms::base_llm::ocr::{error::Error as OcrError, transformation::decode_request_value};
use litellm_llms_types::formats::ocr::OcrDocument;
use rstest::rstest;
use serde_json::{Value, json};
use tower::ServiceExt;
@ -146,7 +147,7 @@ async fn malformed_multipart_uses_an_openai_error_envelope(
#[rstest]
#[case::missing_document(
"/v1/ocr", "mistral/test-ocr", "",
Error::Ocr(OcrDocument::try_from(Value::Null).unwrap_err()),
Error::Ocr(decode_request_value::<OcrDocument>(Value::Null, "document").unwrap_err()),
)]
#[case::empty_document(
"/v1/ocr",

View file

@ -1,15 +1,23 @@
The same ownership rule applies to Messages, Responses, Chat Completions, OCR, and other API formats. This crate owns their shared API data contracts. Adapter contracts and shared transformation machinery belong in `llms/src/base_llm/<format>/`, provider policy in `llms/src/<provider>/<format>/`, and call orchestration in `core/src/<format>/`. A provider originating a format, or several providers using a type, does not change these responsibilities. Existing model locations outside this crate are not exceptions to this rule for new shared API contracts
- `litellm-types` owns shared API data contracts and their serialization
- `litellm-llms-types` owns shared API data contracts and their serialization
- A type belongs here when it describes a request, response, event, or value that consumers must agree on independently of how a call executes
- Being public, serializable, or used by several crates is not sufficient
- These are intended boundaries, not a claim that every existing item follows them
- Organize public contracts by API format: `messages`, `chat_completions`, and `responses`
- Use names such as `litellm_types::messages::MessagesRequest`, without an Anthropic prefix solely because Anthropic designed Messages
- Existing `llms::openai`, `llms::anthropic_messages`, and chat types under `utils` are legacy locations, not patterns for new modules
- Organize public API contracts under `formats`: `messages`, `chat_completions`, `responses`, `ocr`, `audio_transcription`, and `batches`
- Use names such as `litellm_llms_types::formats::messages::MessagesRequest`, without an Anthropic prefix solely because Anthropic designed Messages
- Keep one canonical definition and import path when moving a contract, updating consumers together instead of adding duplicate models or compatibility re-exports
- Keep shared provider-specific wire types and extensions under `providers`
- Provider types may reuse format types; format types must not depend on provider types
- A field belonging to an API format stays under `formats` even when provider support varies. Including it in a type does not promise provider support
- Add a typed provider extension when a consumer needs to interpret or construct it. Keep adapter-only projections in `llms` until a shared public data contract is needed
- Keep one authoritative representation of each field, preserving unknown fields without duplicating typed values in an extension map
- Provider capability checks, defaults, authentication, header selection, and transformations remain in `llms`
- Keep format-independent data helpers such as `headers`, `recognized`, and `serde_compat` at the crate root
- Shared request/response bodies, message and content-block enums, usage records, tool-call chunks, stream-event payloads, and protocol error bodies belong here
- This includes LiteLLM's normalized response contracts and extensions, not just exact upstream schemas
- `ChatCompletionsResponse` currently represents the response handed to the host, so replacing it with a supposedly more complete upstream schema must not silently change that contract
@ -31,6 +39,7 @@ The same ownership rule applies to Messages, Responses, Chat Completions, OCR, a
- Provider config traits, `MessagesTransformContext`, `MessagesModelCapabilities`, `ThinkingBudgets`, `StreamShape`, and transformer state belong in `llms`
- Catalog records and pricing belong in `model-catalog`, which may reuse wire enums such as `ReasoningEffort`
- Host hooks, Python objects, credentials, clients, timeouts, and routing decisions do not become API payload types merely because they cross a crate boundary
- Legacy logging operation selection belongs in `callbacks-legacy-python`, not this crate
- Stream-event data belongs here, but live streams, decoders, framing, buffering, and stream lifecycle decisions do not
- Keep SSE and AWS framing in `framer`, provider decoding and conversion in `llms`, and call orchestration in `core`

View file

@ -1,5 +1,5 @@
[package]
name = "litellm-types"
name = "litellm-llms-types"
version = "0.1.0"
edition.workspace = true
license.workspace = true
@ -9,9 +9,11 @@ repository.workspace = true
schema = ["dep:schemars"]
[dependencies]
macro_rules_attribute.workspace = true
schemars = { workspace = true, optional = true }
serde.workspace = true
serde_json.workspace = true
serde_with.workspace = true
strum.workspace = true
[dev-dependencies]

View file

@ -1,7 +1,6 @@
use serde::{Deserialize, Serialize};
use serde_json::Value;
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
pub struct AudioTranscriptionResponseData {
pub text: String,
}

View file

@ -0,0 +1,36 @@
#[macro_rules_attribute::apply(wire_type)]
#[derive(Copy, Eq)]
#[serde(rename_all = "snake_case")]
pub enum BatchStatus {
InProgress,
Cancelling,
Completed,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Eq)]
pub struct BatchRequestCounts {
pub total: u64,
pub completed: u64,
pub failed: u64,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Eq)]
pub struct BatchResponse {
pub id: String,
pub object: String,
pub endpoint: String,
pub input_file_id: String,
pub completion_window: String,
pub status: BatchStatus,
pub output_file_id: String,
pub created_at: i64,
pub in_progress_at: Option<i64>,
pub expires_at: Option<i64>,
pub completed_at: Option<i64>,
pub expired_at: Option<i64>,
pub cancelling_at: Option<i64>,
pub cancelled_at: Option<i64>,
pub request_counts: BatchRequestCounts,
}

View file

@ -0,0 +1,223 @@
use serde_json::{Map, Value};
use strum::IntoStaticStr;
/// Reasoning effort level accepted or applied by the model.
#[macro_rules_attribute::apply(wire_type)]
#[derive(Copy, Eq, IntoStaticStr)]
#[serde(rename_all = "snake_case")]
#[strum(serialize_all = "snake_case")]
pub enum ReasoningEffort {
None,
Minimal,
Low,
Medium,
High,
Xhigh,
Max,
}
impl ReasoningEffort {
pub const ALL: [Self; 7] = [
Self::None,
Self::Minimal,
Self::Low,
Self::Medium,
Self::High,
Self::Xhigh,
Self::Max,
];
pub fn as_str(self) -> &'static str {
self.into()
}
pub fn parse(value: &str) -> Option<Self> {
Self::ALL
.into_iter()
.find(|effort| effort.as_str() == value)
}
}
#[macro_rules_attribute::apply(wire_type)]
#[serde(untagged)]
pub enum ChatMessageContent {
Text(String),
Parts(Vec<Value>),
}
#[macro_rules_attribute::apply(wire_type)]
pub struct ChatMessage {
pub role: String,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub content: Option<ChatMessageContent>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub name: Option<String>,
#[serde(flatten)]
pub extra: Map<String, Value>,
}
#[macro_rules_attribute::apply(wire_type)]
pub struct ChatCompletionToolCallFunctionChunk {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub name: Option<String>,
pub arguments: String,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub provider_specific_fields: Option<Map<String, Value>>,
}
#[macro_rules_attribute::apply(wire_type)]
pub struct ChatCompletionToolCallChunk {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub id: Option<String>,
#[serde(rename = "type")]
pub tool_type: String,
pub function: ChatCompletionToolCallFunctionChunk,
pub index: i64,
}
#[macro_rules_attribute::apply(wire_type)]
#[serde(tag = "type", rename_all = "snake_case")]
pub enum ChatCompletionThinkingBlock {
Thinking {
#[serde(default, skip_serializing_if = "Option::is_none")]
thinking: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
signature: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
cache_control: Option<Value>,
},
RedactedThinking {
#[serde(default, skip_serializing_if = "Option::is_none")]
data: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
cache_control: Option<Value>,
},
}
/// OpenAI `usage`, including the `prompt_tokens_details` split LiteLLM's Python
/// path reports so cost tracking sees the same numbers on either path.
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct PromptTokensDetails {
pub cached_tokens: u64,
pub cache_creation_tokens: u64,
pub text_tokens: u64,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct ChatCompletionsUsage {
pub prompt_tokens: u64,
pub completion_tokens: u64,
pub total_tokens: u64,
pub prompt_tokens_details: PromptTokensDetails,
}
#[macro_rules_attribute::apply(wire_type)]
pub struct ChatCompletionsChoiceMessage {
pub role: String,
// Whether an empty turn is `None` or `""` is the provider's choice, not a
// shared invariant: Anthropic's transform ends on `merged_text or None`
// while Converse assigns the joined string unconditionally. Each config
// mirrors its own, so keep this optional and serialize it even when None.
pub content: Option<String>,
}
#[macro_rules_attribute::apply(wire_type)]
pub struct ChatCompletionsChoice {
pub index: u64,
pub message: ChatCompletionsChoiceMessage,
pub finish_reason: String,
}
/// The normalized response handed back to the host.
///
/// There is deliberately no `id`: Python mints the `chatcmpl-…` id on the
/// `ModelResponse` it already created, and echoing the provider's own id here
/// would change it. Pinned by `response_carries_no_id` in the Anthropic chat transformation tests.
#[macro_rules_attribute::apply(wire_type)]
pub struct ChatCompletionsResponse {
pub created: u64,
pub model: String,
pub choices: Vec<ChatCompletionsChoice>,
pub usage: ChatCompletionsUsage,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct ChatCompletionDelta {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub content: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub role: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub tool_calls: Option<Vec<ChatCompletionToolCallChunk>>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub reasoning_content: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub thinking_blocks: Option<Vec<ChatCompletionThinkingBlock>>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub provider_specific_fields: Option<Map<String, Value>>,
#[serde(flatten)]
pub extra: Map<String, Value>,
}
#[macro_rules_attribute::apply(wire_type)]
pub struct ChatCompletionStreamingChoice {
pub index: u64,
pub delta: ChatCompletionDelta,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub finish_reason: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub logprobs: Option<Value>,
}
#[macro_rules_attribute::apply(wire_type)]
pub struct ChatCompletionChunk {
pub id: String,
pub created: u64,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub model: Option<String>,
pub object: String,
pub choices: Vec<ChatCompletionStreamingChoice>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub usage: Option<ChatCompletionsUsage>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub provider_specific_fields: Option<Map<String, Value>>,
}
#[cfg(test)]
mod tests {
use rstest::rstest;
use super::*;
#[rstest]
fn reasoning_effort_names_match_the_wire_and_parse_back(
#[values(
ReasoningEffort::None,
ReasoningEffort::Minimal,
ReasoningEffort::Low,
ReasoningEffort::Medium,
ReasoningEffort::High,
ReasoningEffort::Xhigh,
ReasoningEffort::Max
)]
effort: ReasoningEffort,
) {
assert_eq!(
serde_json::to_value(effort).unwrap(),
Value::String(effort.as_str().to_string())
);
assert_eq!(ReasoningEffort::parse(effort.as_str()), Some(effort));
assert!(ReasoningEffort::ALL.contains(&effort));
}
#[rstest]
#[case::unknown("ultra")]
#[case::uppercase("HIGH")]
#[case::empty("")]
fn reasoning_effort_parse_rejects(#[case] value: &str) {
assert_eq!(ReasoningEffort::parse(value), None);
}
}

View file

@ -0,0 +1,11 @@
mod request;
mod response;
pub mod streaming;
pub use request::{
AdaptiveThinking, CacheControl, ContentBlock, ContentBlockType, ContextEdit, ContextManagement,
DisabledThinking, EffortLevel, EnabledThinking, Message, MessageContent,
MessagesOptionalParams, MessagesRequest, MessagesTool, OutputConfig, Speed, SystemPrompt,
ThinkingConfig, ThinkingDisplay,
};
pub use response::MessagesResponse;

View file

@ -1,26 +1,25 @@
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use strum::IntoStaticStr;
use crate::{llms::openai::ReasoningEffort, recognized::Recognized};
use crate::formats::chat_completions::ReasoningEffort;
use crate::recognized::Recognized;
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[serde(untagged)]
pub enum SystemPrompt {
Text(String),
Blocks(Vec<ContentBlock>),
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[serde(untagged)]
pub enum MessageContent {
Text(String),
Blocks(Vec<ContentBlock>),
}
#[derive(
Clone, Debug, PartialEq, Eq, Serialize, Deserialize, strum::Display, strum::EnumString,
)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Eq, strum::Display, strum::EnumString)]
#[serde(from = "String", into = "String")]
#[strum(serialize_all = "snake_case")]
pub enum ContentBlockType {
@ -49,7 +48,8 @@ impl From<ContentBlockType> for String {
}
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct ContentBlock {
#[serde(rename = "type", default, skip_serializing_if = "Option::is_none")]
pub block_type: Option<ContentBlockType>,
@ -93,7 +93,8 @@ impl ContentBlock {
}
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct CacheControl {
#[serde(rename = "type", skip_serializing_if = "Option::is_none")]
pub cache_type: Option<String>,
@ -105,15 +106,16 @@ pub struct CacheControl {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
pub struct AnthropicMessage {
#[macro_rules_attribute::apply(wire_type)]
pub struct Message {
pub role: String,
pub content: MessageContent,
#[serde(flatten)]
pub extra: Map<String, Value>,
}
#[derive(Clone, Copy, Debug, IntoStaticStr, PartialEq, Eq, Hash, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Copy, Hash, IntoStaticStr, Eq)]
#[serde(rename_all = "lowercase")]
#[strum(serialize_all = "lowercase")]
pub enum EffortLevel {
@ -142,7 +144,8 @@ impl From<EffortLevel> for ReasoningEffort {
}
}
#[derive(Clone, Copy, Debug, IntoStaticStr, PartialEq, Eq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Copy, IntoStaticStr, Eq)]
#[serde(rename_all = "lowercase")]
#[strum(serialize_all = "lowercase")]
pub enum Speed {
@ -158,9 +161,9 @@ impl Speed {
/// The tools whose presence changes how the request is sent. Every other tool, custom or
/// server, deserializes as `Recognized::Unrecognized` and passes through verbatim.
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[serde(tag = "type")]
pub enum AnthropicTool {
pub enum MessagesTool {
#[serde(rename = "advisor_20260301")]
Advisor {
#[serde(flatten)]
@ -178,7 +181,7 @@ pub enum AnthropicTool {
},
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[serde(tag = "type")]
pub enum ContextEdit {
#[serde(rename = "compact_20260112")]
@ -198,7 +201,8 @@ pub enum ContextEdit {
},
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct ContextManagement {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub edits: Option<Vec<Recognized<ContextEdit>>>,
@ -206,7 +210,8 @@ pub struct ContextManagement {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct OutputConfig {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub effort: Option<Recognized<EffortLevel>>,
@ -222,7 +227,8 @@ impl OutputConfig {
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Copy, Eq)]
#[serde(rename_all = "lowercase")]
pub enum ThinkingDisplay {
Summarized,
@ -230,7 +236,8 @@ pub enum ThinkingDisplay {
Updates,
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct EnabledThinking {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub budget_tokens: Option<Recognized<u64>>,
@ -240,7 +247,8 @@ pub struct EnabledThinking {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct AdaptiveThinking {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub display: Option<Recognized<ThinkingDisplay>>,
@ -248,13 +256,14 @@ pub struct AdaptiveThinking {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct DisabledThinking {
#[serde(flatten)]
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[serde(tag = "type", rename_all = "lowercase")]
pub enum ThinkingConfig {
Enabled(EnabledThinking),
@ -278,16 +287,17 @@ impl ThinkingConfig {
}
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
pub struct AnthropicMessagesRequest {
#[macro_rules_attribute::apply(wire_type)]
pub struct MessagesRequest {
pub model: String,
pub messages: Vec<AnthropicMessage>,
pub messages: Vec<Message>,
#[serde(flatten)]
pub params: AnthropicMessagesOptionalParams,
pub params: MessagesOptionalParams,
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
pub struct AnthropicMessagesOptionalParams {
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct MessagesOptionalParams {
#[serde(skip_serializing_if = "Option::is_none")]
pub max_tokens: Option<u64>,
#[serde(skip_serializing_if = "Option::is_none")]
@ -305,7 +315,7 @@ pub struct AnthropicMessagesOptionalParams {
#[serde(skip_serializing_if = "Option::is_none")]
pub top_k: Option<i64>,
#[serde(skip_serializing_if = "Option::is_none")]
pub tools: Option<Vec<Recognized<AnthropicTool>>>,
pub tools: Option<Vec<Recognized<MessagesTool>>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub tool_choice: Option<Value>,
#[serde(skip_serializing_if = "Option::is_none")]
@ -334,7 +344,7 @@ pub struct AnthropicMessagesOptionalParams {
pub extra: Map<String, Value>,
}
impl AnthropicMessage {
impl Message {
pub fn blocks(&self) -> &[ContentBlock] {
match &self.content {
MessageContent::Blocks(blocks) => blocks,
@ -357,7 +367,7 @@ mod tests {
use super::*;
fn round_trip<T: serde::de::DeserializeOwned + Serialize>(value: &Value) -> Value {
fn round_trip<T: serde::de::DeserializeOwned + serde::Serialize>(value: &Value) -> Value {
let parsed: T = serde_json::from_value(value.clone()).unwrap();
serde_json::to_value(parsed).unwrap()
}
@ -396,7 +406,7 @@ mod tests {
"stream": true,
"safeguards": [{"type": "dangerous_tool_use"}]
});
let request: AnthropicMessagesRequest = serde_json::from_value(body.clone()).unwrap();
let request: MessagesRequest = serde_json::from_value(body.clone()).unwrap();
assert_eq!(
(
@ -432,7 +442,7 @@ mod tests {
#[case] message: Value,
#[case] expected: Vec<ContentBlock>,
) {
let message: AnthropicMessage = serde_json::from_value(message).unwrap();
let message: Message = serde_json::from_value(message).unwrap();
assert_eq!(message.blocks(), expected.as_slice());
}
@ -440,7 +450,7 @@ mod tests {
#[case::replaces_string_content(json!({"role": "assistant", "content": "old", "name": "kept"}))]
#[case::replaces_block_content(json!({"role": "assistant", "content": [{"type": "text", "text": "old"}], "name": "kept"}))]
fn with_blocks_replaces_content_and_keeps_the_rest(#[case] message: Value) {
let message: AnthropicMessage = serde_json::from_value(message).unwrap();
let message: Message = serde_json::from_value(message).unwrap();
assert_eq!(
serde_json::to_value(message.with_blocks(vec![ContentBlock::text("new")])).unwrap(),
json!({"role": "assistant", "content": [{"type": "text", "text": "new"}], "name": "kept"})
@ -506,7 +516,7 @@ mod tests {
"context_management": [{"type": "compaction", "compact_threshold": 5}]
}))]
fn request_round_trips_unchanged(#[case] request: Value) {
assert_eq!(round_trip::<AnthropicMessagesRequest>(&request), request);
assert_eq!(round_trip::<MessagesRequest>(&request), request);
}
#[rstest]
@ -555,15 +565,15 @@ mod tests {
#[rstest]
#[case::advisor(
json!({"type": "advisor_20260301", "name": "advisor"}),
Recognized::Known(AnthropicTool::Advisor { extra: Map::from_iter([("name".to_string(), json!("advisor"))]) })
Recognized::Known(MessagesTool::Advisor { extra: Map::from_iter([("name".to_string(), json!("advisor"))]) })
)]
#[case::regex_tool_search(
json!({"type": "tool_search_tool_regex_20251119"}),
Recognized::Known(AnthropicTool::ToolSearchRegex { extra: Map::new() })
Recognized::Known(MessagesTool::ToolSearchRegex { extra: Map::new() })
)]
#[case::bm25_tool_search(
json!({"type": "tool_search_tool_bm25_20251119"}),
Recognized::Known(AnthropicTool::ToolSearchBm25 { extra: Map::new() })
Recognized::Known(MessagesTool::ToolSearchBm25 { extra: Map::new() })
)]
#[case::custom_tool_without_a_type(
json!({"name": "advisor", "input_schema": {}}),
@ -576,10 +586,10 @@ mod tests {
#[case::not_an_object(json!("advisor_20260301"), Recognized::Unrecognized(json!("advisor_20260301")))]
fn tools_are_recognized_by_their_exact_type(
#[case] tool: Value,
#[case] expected: Recognized<AnthropicTool>,
#[case] expected: Recognized<MessagesTool>,
) {
assert_eq!(
serde_json::from_value::<Recognized<AnthropicTool>>(tool).unwrap(),
serde_json::from_value::<Recognized<MessagesTool>>(tool).unwrap(),
expected
);
}

View file

@ -1,8 +1,7 @@
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
pub struct AnthropicMessagesResponse {
#[macro_rules_attribute::apply(wire_type)]
pub struct MessagesResponse {
pub id: String,
#[serde(rename = "type")]
pub message_type: String,
@ -31,8 +30,8 @@ mod tests {
stop_sequence: Option<&str>,
usage: Option<Value>,
container: Option<Value>,
) -> AnthropicMessagesResponse {
AnthropicMessagesResponse {
) -> MessagesResponse {
MessagesResponse {
id: "msg_1".to_string(),
message_type: "message".to_string(),
role: "assistant".to_string(),

View file

@ -1,7 +1,7 @@
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct MessagesStreamUsage {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub input_tokens: Option<u64>,
@ -17,7 +17,7 @@ pub struct MessagesStreamUsage {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
pub struct MessagesStreamMessage {
pub id: String,
#[serde(rename = "type")]
@ -32,7 +32,7 @@ pub struct MessagesStreamMessage {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[serde(tag = "type", rename_all = "snake_case")]
pub enum MessagesContentBlockDelta {
TextDelta {
@ -56,7 +56,7 @@ pub enum MessagesContentBlockDelta {
},
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
pub struct MessagesContentBlock {
#[serde(rename = "type")]
pub block_type: String,
@ -82,7 +82,8 @@ pub struct MessagesContentBlock {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct MessagesDelta {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub stop_reason: Option<String>,
@ -96,7 +97,7 @@ pub struct MessagesDelta {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
pub struct MessagesStreamError {
#[serde(rename = "type")]
pub error_type: String,
@ -107,7 +108,7 @@ pub struct MessagesStreamError {
pub extra: Map<String, Value>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[serde(tag = "type", rename_all = "snake_case")]
pub enum MessagesStreamEvent {
MessageStart {

View file

@ -0,0 +1,6 @@
pub mod audio_transcription;
pub mod batches;
pub mod chat_completions;
pub mod messages;
pub mod ocr;
pub mod responses;

View file

@ -0,0 +1,152 @@
use std::collections::BTreeMap;
use serde_json::{Map, Value};
use serde_with::serde_as;
use crate::serde_compat::{FiniteF64, LaxI64};
#[macro_rules_attribute::apply(wire_type)]
#[serde(tag = "type")]
pub enum OcrDocument {
#[serde(rename = "document_url")]
DocumentUrl {
document_url: String,
#[serde(flatten)]
extra_fields: BTreeMap<String, Option<String>>,
},
#[serde(rename = "image_url")]
ImageUrl {
image_url: String,
#[serde(flatten)]
extra_fields: BTreeMap<String, Option<String>>,
},
}
impl OcrDocument {
pub fn source(&self) -> &str {
match self {
Self::DocumentUrl { document_url, .. } => document_url,
Self::ImageUrl { image_url, .. } => image_url,
}
}
pub fn is_remote(&self) -> bool {
let source = self.source();
source.starts_with("http://") || source.starts_with("https://")
}
pub fn with_source(self, source: String) -> Self {
match self {
Self::DocumentUrl { extra_fields, .. } => Self::DocumentUrl {
document_url: source,
extra_fields,
},
Self::ImageUrl { extra_fields, .. } => Self::ImageUrl {
image_url: source,
extra_fields,
},
}
}
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Copy, Default, Eq)]
#[serde(rename_all = "lowercase")]
pub enum OcrResponseFormat {
#[default]
Litellm,
Native,
}
#[serde_as]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct OcrPageDimensions {
#[serde_as(deserialize_as = "Option<LaxI64>")]
pub dpi: Option<i64>,
#[serde_as(deserialize_as = "Option<LaxI64>")]
pub height: Option<i64>,
#[serde_as(deserialize_as = "Option<LaxI64>")]
pub width: Option<i64>,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct OcrPageImage {
pub image_base64: Option<String>,
pub bbox: Option<Map<String, Value>>,
#[serde(flatten)]
pub extra_fields: Map<String, Value>,
}
#[serde_as]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct OcrPage {
#[serde_as(deserialize_as = "LaxI64")]
pub index: i64,
pub markdown: String,
pub images: Option<Vec<OcrPageImage>>,
pub dimensions: Option<OcrPageDimensions>,
#[serde(flatten)]
pub extra_fields: Map<String, Value>,
}
#[serde_as]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct OcrUsageInfo {
#[serde_as(deserialize_as = "Option<LaxI64>")]
pub pages_processed: Option<i64>,
#[serde_as(deserialize_as = "Option<LaxI64>")]
pub pages_processed_annotation: Option<i64>,
#[serde_as(deserialize_as = "Option<FiniteF64>")]
pub credits: Option<f64>,
#[serde_as(deserialize_as = "Option<LaxI64>")]
pub doc_size_bytes: Option<i64>,
#[serde(flatten)]
pub extra_fields: Map<String, Value>,
}
#[macro_rules_attribute::apply(wire_type)]
pub struct LiteLLMOcrResponse {
pub pages: Vec<OcrPage>,
pub model: String,
pub document_annotation: Option<Value>,
pub usage_info: Option<OcrUsageInfo>,
pub content: Option<String>,
pub tables: Option<Vec<Map<String, Value>>>,
#[serde(rename = "keyValuePairs")]
pub key_value_pairs: Option<Vec<Map<String, Value>>>,
#[serde(default = "ocr_object")]
pub object: String,
#[serde(flatten)]
pub extra_fields: Map<String, Value>,
#[serde(skip_serializing_if = "Option::is_none")]
pub provider_native_response: Option<Map<String, Value>>,
}
impl LiteLLMOcrResponse {
pub fn new(model: impl Into<String>, pages: Vec<OcrPage>) -> Self {
Self {
pages,
model: model.into(),
document_annotation: None,
usage_info: None,
content: None,
tables: None,
key_value_pairs: None,
object: ocr_object(),
extra_fields: Map::new(),
provider_native_response: None,
}
}
pub fn into_json(self) -> Value {
serde_json::to_value(self).expect("OCR response fields are JSON-compatible")
}
}
fn ocr_object() -> String {
"ocr".into()
}

View file

@ -0,0 +1,4 @@
mod response;
pub mod streaming_websocket;
pub use response::ResponsesApiResponse;

View file

@ -1,7 +1,6 @@
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
pub struct ResponsesApiResponse {
pub id: String,
pub model: String,

View file

@ -2,6 +2,8 @@ use serde::{Deserialize, Deserializer, Serialize, Serializer};
use serde_json::{Map, Value};
#[derive(Clone, Debug, PartialEq, Eq, strum::AsRefStr)]
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
#[cfg_attr(feature = "schema", schemars(with = "String"))]
pub enum ResponsesWsEventType {
#[strum(serialize = "response.create")]
ResponseCreate,
@ -52,7 +54,7 @@ impl<'de> Deserialize<'de> for ResponsesWsEventType {
}
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
pub struct ResponsesWsEvent {
#[serde(rename = "type")]
pub event_type: ResponsesWsEventType,
@ -78,7 +80,8 @@ impl ResponsesWsEvent {
}
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Eq)]
pub struct ResponsesErrorFrame {
#[serde(rename = "type")]
pub frame_type: &'static str,
@ -97,7 +100,8 @@ impl ResponsesErrorFrame {
}
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Eq)]
pub struct ResponsesErrorBody {
#[serde(rename = "type")]
pub error_type: &'static str,

View file

@ -0,0 +1,17 @@
use serde_json::{Map, Value};
#[macro_rules_attribute::apply(wire_type)]
#[derive(Default)]
pub struct ProviderSpecificHeader {
#[serde(default)]
pub custom_llm_provider: String,
#[serde(default)]
pub extra_headers: Map<String, Value>,
}
#[macro_rules_attribute::apply(wire_type)]
#[serde(untagged)]
pub enum ProviderSpecificHeaders {
One(ProviderSpecificHeader),
Many(Vec<ProviderSpecificHeader>),
}

View file

@ -0,0 +1,11 @@
macro_rules_attribute::attribute_alias! {
#[apply(wire_type)] =
#[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)]
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
}
pub mod formats;
pub mod headers;
pub mod providers;
pub mod recognized;
pub mod serde_compat;

View file

@ -0,0 +1 @@
pub mod anthropic;

View file

@ -1,7 +1,6 @@
use serde::{Deserialize, Serialize};
use serde_json::Value;
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[macro_rules_attribute::apply(wire_type)]
#[serde(untagged)]
pub enum Recognized<T> {
Known(T),

View file

@ -0,0 +1,113 @@
use serde::{
Deserializer,
de::{Error, Visitor},
};
use serde_with::DeserializeAs;
pub struct LaxI64;
pub struct FiniteF64;
impl<'de> DeserializeAs<'de, i64> for LaxI64 {
fn deserialize_as<D: Deserializer<'de>>(deserializer: D) -> Result<i64, D::Error> {
deserializer.deserialize_any(Self)
}
}
impl<'de> Visitor<'de> for LaxI64 {
type Value = i64;
fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
formatter.write_str("an integer in the i64 range")
}
fn visit_i64<E: Error>(self, value: i64) -> Result<i64, E> {
Ok(value)
}
fn visit_u64<E: Error>(self, value: u64) -> Result<i64, E> {
i64::try_from(value).map_err(E::custom)
}
fn visit_f64<E: Error>(self, value: f64) -> Result<i64, E> {
integral_float(value).ok_or_else(|| E::custom("expected an integer in the i64 range"))
}
fn visit_str<E: Error>(self, value: &str) -> Result<i64, E> {
integer_string(value.trim())
.ok_or_else(|| E::custom("expected an integer in the i64 range"))
}
fn visit_bool<E: Error>(self, value: bool) -> Result<i64, E> {
Ok(i64::from(value))
}
}
impl<'de> DeserializeAs<'de, f64> for FiniteF64 {
fn deserialize_as<D: Deserializer<'de>>(deserializer: D) -> Result<f64, D::Error> {
deserializer.deserialize_any(Self)
}
}
impl<'de> Visitor<'de> for FiniteF64 {
type Value = f64;
fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
formatter.write_str("a finite number")
}
fn visit_i64<E: Error>(self, value: i64) -> Result<f64, E> {
Ok(value as f64)
}
fn visit_u64<E: Error>(self, value: u64) -> Result<f64, E> {
Ok(value as f64)
}
fn visit_f64<E: Error>(self, value: f64) -> Result<f64, E> {
value
.is_finite()
.then_some(value)
.ok_or_else(|| E::custom("expected a finite number"))
}
fn visit_str<E: Error>(self, value: &str) -> Result<f64, E> {
self.visit_f64(value.trim().parse::<f64>().map_err(E::custom)?)
}
fn visit_bool<E: Error>(self, value: bool) -> Result<f64, E> {
Ok(f64::from(value))
}
}
fn integer_string(value: &str) -> Option<i64> {
let integer = match value.split_once('.') {
Some((integer, fraction)) => {
if fraction.is_empty() || !fraction.bytes().all(|byte| byte == b'0') {
return None;
}
integer
}
None => value,
};
if integer.starts_with('_') || integer.ends_with('_') || integer.contains("__") {
return None;
}
let digits = integer.strip_prefix(['+', '-']).unwrap_or(integer);
if digits.is_empty()
|| digits.starts_with('_')
|| !digits
.bytes()
.all(|byte| byte.is_ascii_digit() || byte == b'_')
{
return None;
}
integer.replace('_', "").parse().ok()
}
fn integral_float(value: f64) -> Option<i64> {
(value.is_finite()
&& value.fract() == 0.0
&& value >= i64::MIN as f64
&& value < -(i64::MIN as f64))
.then_some(value as i64)
}

View file

@ -1,4 +1,4 @@
use litellm_types::llms::anthropic_messages::anthropic_request::{ContentBlock, ContentBlockType};
use litellm_llms_types::formats::messages::{ContentBlock, ContentBlockType};
use rstest::rstest;
use serde_json::{Value, json};

View file

@ -1,4 +1,4 @@
use litellm_types::messages::streaming::MessagesStreamEvent;
use litellm_llms_types::formats::messages::streaming::MessagesStreamEvent;
use rstest::rstest;
use serde_json::{Value, json};

View file

@ -0,0 +1,104 @@
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument, OcrPage};
use rstest::rstest;
use serde_json::{Map, Value, json};
#[rstest]
#[case::missing_page_fields(json!({"pages": [{}]}))]
#[case::invalid_markdown(json!({"pages": [{"index": 0, "markdown": false}]}))]
#[case::invalid_image_bounds(json!({"pages": [{"index": 0, "markdown": "", "images": [{"bbox": []}]}]}))]
#[case::fractional_page_count(json!({"usage_info": {"pages_processed": 1.5}}))]
#[case::invalid_table(json!({"tables": [false]}))]
#[case::invalid_key_value_pair(json!({"keyValuePairs": [[]]}))]
#[case::invalid_native_response(json!({"provider_native_response": []}))]
fn normalized_response_rejects_invalid_shared_fields(#[case] fields: Value) {
let payload: Map<String, Value> = json!({"model": "model", "pages": []})
.as_object()
.unwrap()
.iter()
.chain(fields.as_object().unwrap())
.map(|(key, value)| (key.clone(), value.clone()))
.collect();
assert!(serde_json::from_value::<LiteLLMOcrResponse>(Value::Object(payload)).is_err());
}
#[rstest]
fn document_rejects_non_string_provider_fields() {
assert!(
serde_json::from_value::<OcrDocument>(json!({
"type": "image_url", "image_url": "https://example.com/image", "detail": 42
}))
.is_err()
);
}
#[rstest]
#[case::large_integer(json!("9007199254740993.0"), 9_007_199_254_740_993)]
#[case::signed_decimal(json!("+2.000"), 2)]
#[case::separator(json!("1_000"), 1000)]
#[case::boolean(json!(true), 1)]
#[case::integral_float(json!(2.0), 2)]
fn numeric_coercion_preserves_integer_precision(#[case] value: Value, #[case] expected: i64) {
let page: OcrPage = serde_json::from_value(json!({"index": value, "markdown": ""})).unwrap();
assert_eq!(page.index, expected);
assert_eq!(
serde_json::to_value(page).unwrap()["index"],
json!(expected)
);
}
#[rstest]
#[case::exponent(json!("1e2"))]
#[case::missing_integer(json!(".0"))]
#[case::missing_fraction(json!("2."))]
#[case::leading_separator(json!("_2"))]
#[case::repeated_separator(json!("2__0"))]
#[case::fractional_float(json!(2.5))]
#[case::null(json!(null))]
fn page_index_rejects_invalid_integers(#[case] value: Value) {
assert!(serde_json::from_value::<OcrPage>(json!({"index": value, "markdown": ""})).is_err());
}
#[rstest]
#[case::document_url("document_url", "document_name", "application/pdf")]
#[case::image_url("image_url", "detail", "image/png")]
fn document_variants_preserve_provider_fields_when_rewriting_sources(
#[case] kind: &str,
#[case] field: &str,
#[case] mime_type: &str,
#[values(json!("kept"), Value::Null)] extra: Value,
) {
let original = "https://example.com/input";
let replacement = format!("data:{mime_type};base64,AA==");
let document: OcrDocument =
serde_json::from_value(json!({"type": kind, kind: original, field: extra})).unwrap();
assert_eq!(document.source(), original);
assert!(document.is_remote());
let rewritten = document.with_source(replacement.clone());
assert!(!rewritten.is_remote());
assert_eq!(
serde_json::to_value(rewritten).unwrap(),
json!({"type": kind, kind: replacement, field: extra})
);
}
#[rstest]
#[case::absent_native(None)]
#[case::present_native(Some(Map::from_iter([("native".into(), json!({"nested": [null, 1]}))])))]
fn response_serialization_preserves_extensions_and_native_presence(
#[case] native: Option<Map<String, Value>>,
) {
let response = LiteLLMOcrResponse {
extra_fields: Map::from_iter([("provider_field".into(), json!("kept"))]),
provider_native_response: native.clone(),
..LiteLLMOcrResponse::new("model", vec![])
};
let serialized = response.into_json();
assert_eq!(serialized["provider_field"], "kept");
assert_eq!(
serialized.get("provider_native_response").cloned(),
native.clone().map(Value::Object)
);
let decoded: LiteLLMOcrResponse = serde_json::from_value(serialized.clone()).unwrap();
assert_eq!(decoded.provider_native_response, native);
assert_eq!(decoded.into_json(), serialized);
}

View file

@ -0,0 +1,86 @@
use litellm_llms_types::serde_compat::{FiniteF64, LaxI64};
use rstest::rstest;
use serde::{Deserialize, Serialize};
use serde_json::{Value, json};
use serde_with::serde_as;
#[serde_as]
#[derive(Debug, Deserialize, Serialize, PartialEq)]
struct Numbers {
#[serde_as(deserialize_as = "Option<Vec<LaxI64>>")]
integers: Option<Vec<i64>>,
#[serde_as(deserialize_as = "Option<FiniteF64>")]
float: Option<f64>,
}
#[rstest]
fn adapters_compose_and_serialize_as_numbers() {
let numbers: Numbers = serde_json::from_value(json!({
"integers": ["9007199254740993.0", "1_000", " +2.000 ", 3.0, true],
"float": " 1.5 "
}))
.unwrap();
assert_eq!(
serde_json::to_value(numbers).unwrap(),
json!({"integers": [9_007_199_254_740_993_i64, 1000, 2, 3, 1], "float": 1.5})
);
}
#[rstest]
#[case::missing(json!({}))]
#[case::null(json!({"integers": null, "float": null}))]
fn optional_adapters_accept_missing_and_null_fields(#[case] input: Value) {
assert_eq!(
serde_json::from_value::<Numbers>(input).unwrap(),
Numbers {
integers: None,
float: None
}
);
}
#[rstest]
#[case::minimum(json!(i64::MIN), i64::MIN)]
#[case::maximum(json!(i64::MAX), i64::MAX)]
#[case::maximum_string(json!(i64::MAX.to_string()), i64::MAX)]
fn integers_preserve_bounds(#[case] input: Value, #[case] expected: i64) {
let numbers: Numbers = serde_json::from_value(json!({"integers": [input]})).unwrap();
assert_eq!(numbers.integers, Some(vec![expected]));
}
#[rstest]
#[case::unsigned_maximum(json!(u64::MAX))]
#[case::above_maximum(json!(9_223_372_036_854_775_808_u64))]
#[case::float_above_maximum(json!(9_223_372_036_854_775_808.0))]
#[case::below_minimum(json!("-9223372036854775809"))]
#[case::precise_fraction(json!("1.0000000000000001"))]
#[case::exponent(json!("1e3"))]
#[case::missing_fraction(json!("2."))]
#[case::missing_integer(json!(".0"))]
#[case::leading_separator(json!("_2"))]
#[case::repeated_separator(json!("2__0"))]
#[case::fraction(json!(2.5))]
#[case::null(json!(null))]
#[case::object(json!({}))]
fn integers_reject_invalid_values(#[case] input: Value) {
assert!(serde_json::from_value::<Numbers>(json!({"integers": [input]})).is_err());
}
#[rstest]
#[case::nan(json!("NaN"))]
#[case::positive_infinity(json!("inf"))]
#[case::negative_infinity(json!("-inf"))]
#[case::overflow(json!("1e999"))]
#[case::array(json!([]))]
fn floats_reject_nonfinite_and_invalid_values(#[case] input: Value) {
assert!(serde_json::from_value::<Numbers>(json!({"float": input})).is_err());
}
#[rstest]
#[case::integer(json!(2), 2.0)]
#[case::float(json!(2.5), 2.5)]
#[case::boolean(json!(true), 1.0)]
fn floats_accept_finite_numbers(#[case] input: Value, #[case] expected: f64) {
let numbers: Numbers = serde_json::from_value(json!({"float": input})).unwrap();
assert_eq!(numbers.float, Some(expected));
}

View file

@ -0,0 +1,32 @@
use litellm_llms_types::formats::chat_completions::ChatMessage;
use rstest::rstest;
use serde_json::json;
#[rstest]
fn wire_type_preserves_serialization() {
let message = ChatMessage {
role: "user".to_owned(),
content: None,
name: None,
extra: Default::default(),
};
assert_eq!(
serde_json::to_value(message).unwrap(),
json!({"role": "user"})
);
}
#[cfg(feature = "schema")]
#[rstest]
fn wire_type_supports_schema_generation() {
let schema = schemars::schema_for!(ChatMessage);
assert!(
schema
.to_value()
.get("properties")
.and_then(serde_json::Value::as_object)
.is_some_and(|properties| properties.contains_key("role"))
);
}

View file

@ -12,7 +12,7 @@ Use trait defaults for unchanged inherited behavior and explicit delegation for
Use named `#[rstest]` cases for independent input/output scenarios instead of loops or repeated calls in one test. Inject reusable setup with `#[fixture]` arguments and use `#[with(...)]` for fixture overrides. Keep assertions about the same result together
Base OCR currently keeps response models next to `BaseOcrConfig` in `src/base_llm/ocr/transformation.rs`. This is legacy placement, not an exception to the shared API contract ownership in `litellm-types`. Rust context/environment types support the runtime. `BaseOcrConfig::prepare_request` corresponds to Python's HTTP-handler preparation rather than a `BaseOCRConfig` method, and `validate_request_body` is a Rust-only hook. `src/base_llm/ocr/error.rs` and `src/base_llm/ocr/document.rs` are Rust-only: the OCR error taxonomy shared with the route, and inline-document helpers shared by several providers
Shared OCR document and response contracts live in `litellm-llms-types::formats::ocr`. `BaseOcrConfig` and decoding into adapter errors remain in `src/base_llm/ocr/transformation.rs`. Rust context/environment types support the runtime. `BaseOcrConfig::prepare_request` corresponds to Python's HTTP-handler preparation rather than a `BaseOCRConfig` method, and `validate_request_body` is a Rust-only hook. `src/base_llm/ocr/error.rs` and `src/base_llm/ocr/document.rs` are Rust-only: the OCR error taxonomy shared with the route, and inline-document helpers shared by several providers
For Mistral, `async_transform_ocr_request` uses the base default in both languages. `resolve_headers` and `build_ocr_url` implement the respective environment and URL operations, and `normalize_response` implements the typed part of response transformation. Existing auth key/header handling and top-level response-extra preservation differ between languages; layout refactors must preserve those behaviors and verify them with the existing tests
@ -22,7 +22,7 @@ Azure Messages maps to `llms/azure_ai/anthropic/messages_transformation.py`; Bed
## Provider and format boundaries
The same ownership rule applies to Messages, Responses, Chat Completions, OCR, and other API formats. `litellm-types` owns shared API data contracts. `llms/src/base_llm/<format>/` owns provider adapter contracts and shared transformation machinery. `llms/src/<provider>/<format>/` owns provider implementations and policy. `core/src/<format>/` owns call orchestration. Repeating a format name identifies the API each layer handles, not duplicate ownership of its schema. These boundaries also apply between modules in the same crate
The same ownership rule applies to Messages, Responses, Chat Completions, OCR, and other API formats. `litellm-llms-types` owns shared API data contracts. `llms/src/base_llm/<format>/` owns provider adapter contracts and shared transformation machinery. `llms/src/<provider>/<format>/` owns provider implementations and policy. `core/src/<format>/` owns call orchestration. Repeating a format name identifies the API each layer handles, not duplicate ownership of its schema. These boundaries also apply between modules in the same crate
A provider adapter may explicitly reuse another provider's transformation helper when that policy applies to its backend, such as Bedrock's Claude adapter using Anthropic payload shaping. Reuse across hosts of the same model family does not make the policy format-wide. Keep provider policy out of shared trait defaults and generic normalization, and keep shared execution contexts limited to inputs the adapter contract actually needs. Pure payload rewrites belong with transformations, not transport handlers

View file

@ -9,7 +9,7 @@ repository.workspace = true
test-support = ["litellm-http/test-support"]
[dependencies]
litellm-types.workspace = true
litellm-llms-types.workspace = true
litellm-core-utils.workspace = true
litellm-auth = { workspace = true, features = ["aws", "azure", "gcp"] }
litellm-auth-aws.workspace = true

View file

@ -3,6 +3,6 @@
- Put behavior specific to the Messages API in `messages/`
- Keep generic HTTP mechanics in `litellm-http`, configuration lookup in the existing settings utilities, and credential application in the shared auth layer
- Choose authentication policy and required headers here, then let shared infrastructure apply those decisions
- Consume shared API contracts from `litellm-types`. Do not define public Messages protocol types under this provider
- Consume shared API contracts from `litellm-llms-types`. Do not define public Messages protocol types under this provider
- Preserve Python's concepts and observable behavior where useful, without mechanically reproducing its class hierarchy, helpers, or file structure
- `ReplayedWebSearchResult` and `ReplayedWebSearchContent` are private partial models for replay flattening, not complete public protocol contracts. Keep them private while they serve that transformation

View file

@ -1,4 +1,5 @@
use litellm_types::llms::anthropic_messages::anthropic_response::AnthropicMessagesResponse;
use litellm_llms_types::formats::batches::{BatchRequestCounts, BatchResponse, BatchStatus};
use litellm_llms_types::formats::messages::MessagesResponse;
use serde::{Deserialize, Serialize};
use serde_json::Value;
use time::OffsetDateTime;
@ -45,46 +46,8 @@ struct BatchResultRecord {
#[derive(Deserialize)]
#[serde(tag = "type", rename_all = "snake_case")]
enum BatchResult {
Succeeded {
message: Box<AnthropicMessagesResponse>,
},
Errored {
error: Value,
},
}
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum BatchStatus {
InProgress,
Cancelling,
Completed,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct BatchRequestCounts {
pub total: u64,
pub completed: u64,
pub failed: u64,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct LiteLlmMessageBatch {
pub id: String,
pub object: String,
pub endpoint: String,
pub input_file_id: String,
pub completion_window: String,
pub status: BatchStatus,
pub output_file_id: String,
pub created_at: i64,
pub in_progress_at: Option<i64>,
pub expires_at: Option<i64>,
pub completed_at: Option<i64>,
pub expired_at: Option<i64>,
pub cancelling_at: Option<i64>,
pub cancelled_at: Option<i64>,
pub request_counts: BatchRequestCounts,
Succeeded { message: Box<MessagesResponse> },
Errored { error: Value },
}
pub trait AnthropicBatchesConfig {
@ -100,7 +63,7 @@ pub trait AnthropicBatchesConfig {
&self,
response: AnthropicMessageBatch,
now: i64,
) -> Result<LiteLlmMessageBatch, Error>;
) -> Result<BatchResponse, Error>;
fn retrieve_batch_url(
&self,
@ -115,9 +78,9 @@ pub trait AnthropicBatchesConfig {
&self,
response: AnthropicMessageBatch,
now: i64,
) -> LiteLlmMessageBatch;
) -> BatchResponse;
fn transform_batch_results(&self, body: &str) -> Result<Vec<AnthropicMessagesResponse>, Error>;
fn transform_batch_results(&self, body: &str) -> Result<Vec<MessagesResponse>, Error>;
}
pub struct AnthropicBatchesTransformation;
@ -172,7 +135,7 @@ impl AnthropicBatchesConfig for AnthropicBatchesTransformation {
&self,
_response: AnthropicMessageBatch,
_now: i64,
) -> Result<LiteLlmMessageBatch, Error> {
) -> Result<BatchResponse, Error> {
Err(Error::Unsupported("Anthropic message batch creation"))
}
@ -200,7 +163,7 @@ impl AnthropicBatchesConfig for AnthropicBatchesTransformation {
&self,
response: AnthropicMessageBatch,
now: i64,
) -> LiteLlmMessageBatch {
) -> BatchResponse {
let created_at = timestamp(response.created_at.as_deref());
let ended_at = timestamp(response.ended_at.as_deref());
let expires_at = timestamp(response.expires_at.as_deref());
@ -221,7 +184,7 @@ impl AnthropicBatchesConfig for AnthropicBatchesTransformation {
failed: response.request_counts.errored,
};
LiteLlmMessageBatch {
BatchResponse {
id: response.id.clone(),
object: "batch".into(),
endpoint: "/v1/messages".into(),
@ -248,7 +211,7 @@ impl AnthropicBatchesConfig for AnthropicBatchesTransformation {
}
}
fn transform_batch_results(&self, body: &str) -> Result<Vec<AnthropicMessagesResponse>, Error> {
fn transform_batch_results(&self, body: &str) -> Result<Vec<MessagesResponse>, Error> {
body.lines()
.filter(|line| !line.trim().is_empty())
.enumerate()

View file

@ -1,11 +1,13 @@
use std::collections::HashMap;
use litellm_types::messages::streaming::{
MessagesContentBlock, MessagesContentBlockDelta, MessagesStreamEvent, MessagesStreamUsage,
};
use litellm_types::{
llms::openai::{ChatCompletionThinkingBlock, ChatCompletionToolCallChunk},
utils::{ChatCompletionChunk, ChatCompletionsUsage},
use litellm_llms_types::formats::{
chat_completions::{
ChatCompletionChunk, ChatCompletionThinkingBlock, ChatCompletionToolCallChunk,
ChatCompletionsUsage,
},
messages::streaming::{
MessagesContentBlock, MessagesContentBlockDelta, MessagesStreamEvent, MessagesStreamUsage,
},
};
use serde_json::Value;

View file

@ -3,9 +3,8 @@ use litellm_core_utils::{
core_helpers::{finish_reason_for, unix_now, usage_from_parts},
prompt_templates::factory::{Conversation, build_conversation},
};
use litellm_types::{
llms::openai::ChatMessage,
utils::{ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse},
use litellm_llms_types::formats::chat_completions::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
};
use serde::Deserialize;
use serde_json::{Map, Value, json};
@ -50,16 +49,16 @@ const SUPPORTED_PARAMS: &[(&str, &str)] = &[
];
#[derive(Deserialize)]
struct MessageResponse {
struct TextResponseProjection {
model: String,
content: Vec<ContentBlock>,
usage: MessageUsage,
content: Vec<TextResponseBlock>,
usage: ResponseUsageProjection,
stop_reason: Option<String>,
}
#[derive(Deserialize)]
#[serde(tag = "type", rename_all = "snake_case")]
enum ContentBlock {
enum TextResponseBlock {
Text {
text: String,
},
@ -68,7 +67,7 @@ enum ContentBlock {
}
#[derive(Deserialize)]
struct MessageUsage {
struct ResponseUsageProjection {
input_tokens: u64,
output_tokens: u64,
#[serde(default)]
@ -124,16 +123,17 @@ impl BaseConfig for AnthropicConfig {
_model: &str,
response: ProviderChatResponseData,
) -> Result<ChatCompletionsResponse, Error> {
let body: MessageResponse = serde_json::from_value(response.body).map_err(|error| {
Error::InvalidResponse(crate::ErrorDetail::invalid("messages response", error))
})?;
let body: TextResponseProjection =
serde_json::from_value(response.body).map_err(|error| {
Error::InvalidResponse(crate::ErrorDetail::invalid("messages response", error))
})?;
// The route declines tool and thinking requests, so a non-text block
// means the response carries something this path never asked for.
// Decline rather than silently dropping it; the host falls back.
if body
.content
.iter()
.any(|block| matches!(block, ContentBlock::Other))
.any(|block| matches!(block, TextResponseBlock::Other))
{
return Err(Error::Unsupported("non-text response content block"));
}
@ -141,8 +141,8 @@ impl BaseConfig for AnthropicConfig {
.content
.into_iter()
.map(|block| match block {
ContentBlock::Text { text } => text,
ContentBlock::Other => String::new(),
TextResponseBlock::Text { text } => text,
TextResponseBlock::Other => String::new(),
})
.collect();

View file

@ -4,14 +4,13 @@ use litellm_core_utils::settings::resolve_non_empty;
use litellm_http::request::{
has_header, header_value, header_values, with_header, without_headers,
};
use litellm_types::llms::{
anthropic::{AnthropicBeta, BetaSet},
anthropic_messages::anthropic_request::{
AnthropicMessage, AnthropicTool, ContentBlock, ContentBlockType, EffortLevel,
MessageContent,
use litellm_llms_types::{
formats::messages::{
ContentBlock, ContentBlockType, EffortLevel, Message, MessageContent, MessagesTool,
},
providers::anthropic::{AnthropicBeta, BetaSet},
recognized::Recognized,
};
use litellm_types::recognized::Recognized;
use serde::Deserialize;
use serde_json::Value;
@ -229,36 +228,30 @@ pub fn optionally_handle_anthropic_oauth(headers: Headers, api_key: Option<&str>
OauthHandling::Untouched(headers)
}
pub fn is_tool_search_used(tools: Option<&[Recognized<AnthropicTool>]>) -> bool {
pub fn is_tool_search_used(tools: Option<&[Recognized<MessagesTool>]>) -> bool {
tools.into_iter().flatten().any(|tool| {
matches!(
tool,
Recognized::Known(
AnthropicTool::ToolSearchRegex { .. } | AnthropicTool::ToolSearchBm25 { .. }
MessagesTool::ToolSearchRegex { .. } | MessagesTool::ToolSearchBm25 { .. }
)
)
})
}
pub fn has_advisor_tool(tools: Option<&[Recognized<AnthropicTool>]>) -> bool {
pub fn has_advisor_tool(tools: Option<&[Recognized<MessagesTool>]>) -> bool {
tools
.into_iter()
.flatten()
.any(|tool| matches!(tool, Recognized::Known(AnthropicTool::Advisor { .. })))
.any(|tool| matches!(tool, Recognized::Known(MessagesTool::Advisor { .. })))
}
pub fn requires_native_compaction_beta(
compaction: Option<&Value>,
messages: &[AnthropicMessage],
) -> bool {
pub fn requires_native_compaction_beta(compaction: Option<&Value>, messages: &[Message]) -> bool {
compaction.is_some()
|| messages
.iter()
.flat_map(AnthropicMessage::blocks)
.any(|block| {
block.is_type(ContentBlockType::Compaction)
&& block.signature.as_deref().is_some_and(|s| !s.is_empty())
})
|| messages.iter().flat_map(Message::blocks).any(|block| {
block.is_type(ContentBlockType::Compaction)
&& block.signature.as_deref().is_some_and(|s| !s.is_empty())
})
}
fn is_blank(text: Option<&str>) -> bool {
@ -273,10 +266,7 @@ pub fn is_empty_thinking_block(block: &ContentBlock) -> bool {
block.is_type(ContentBlockType::Thinking) && is_blank(block.thinking.as_deref())
}
fn retain_blocks(
messages: Vec<AnthropicMessage>,
keep: impl Fn(&ContentBlock) -> bool,
) -> Vec<AnthropicMessage> {
fn retain_blocks(messages: Vec<Message>, keep: impl Fn(&ContentBlock) -> bool) -> Vec<Message> {
messages
.into_iter()
.filter_map(|message| match message.content {
@ -293,7 +283,7 @@ fn retain_blocks(
.collect()
}
pub fn strip_empty_content_blocks(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
pub fn strip_empty_content_blocks(messages: Vec<Message>) -> Vec<Message> {
retain_blocks(messages, |block| {
!is_empty_text_block(block) && !is_empty_thinking_block(block)
})
@ -350,11 +340,11 @@ fn sanitize_tool_use_id_block(block: ContentBlock) -> ContentBlock {
}
}
pub fn sanitize_tool_use_ids(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
pub fn sanitize_tool_use_ids(messages: Vec<Message>) -> Vec<Message> {
messages
.into_iter()
.map(|message| match message.content {
MessageContent::Blocks(blocks) => AnthropicMessage {
MessageContent::Blocks(blocks) => Message {
content: MessageContent::Blocks(
blocks.into_iter().map(sanitize_tool_use_id_block).collect(),
),
@ -365,11 +355,11 @@ pub fn sanitize_tool_use_ids(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMe
.collect()
}
pub fn strip_provider_specific_fields(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
pub fn strip_provider_specific_fields(messages: Vec<Message>) -> Vec<Message> {
messages
.into_iter()
.map(|message| match message.content {
MessageContent::Blocks(blocks) => AnthropicMessage {
MessageContent::Blocks(blocks) => Message {
content: MessageContent::Blocks(
blocks
.into_iter()
@ -395,7 +385,7 @@ pub fn is_encrypted_reasoning_block(block: &ContentBlock) -> bool {
field.is_some_and(|value| value.starts_with(ENCRYPTED_REASONING_SIGNATURE_PREFIX))
}
pub fn strip_encrypted_reasoning_blocks(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
pub fn strip_encrypted_reasoning_blocks(messages: Vec<Message>) -> Vec<Message> {
retain_blocks(messages, |block| !is_encrypted_reasoning_block(block))
}
@ -405,7 +395,7 @@ fn is_advisor_use(block: &ContentBlock) -> bool {
&& block.id.as_deref().is_some_and(|id| !id.is_empty())
}
pub fn strip_advisor_blocks(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
pub fn strip_advisor_blocks(messages: Vec<Message>) -> Vec<Message> {
messages
.into_iter()
.map(|message| {
@ -588,13 +578,11 @@ fn flatten_web_search_results_in_blocks(blocks: Vec<ContentBlock>) -> Vec<Conten
.collect()
}
pub fn flatten_unencrypted_web_search_results(
messages: Vec<AnthropicMessage>,
) -> Vec<AnthropicMessage> {
pub fn flatten_unencrypted_web_search_results(messages: Vec<Message>) -> Vec<Message> {
messages
.into_iter()
.map(|message| match message.content {
MessageContent::Blocks(blocks) => AnthropicMessage {
MessageContent::Blocks(blocks) => Message {
content: MessageContent::Blocks(flatten_web_search_results_in_blocks(blocks)),
..message
},
@ -619,11 +607,8 @@ mod tests {
EffortLevel::Max,
];
fn apply(
sanitizer: fn(Vec<AnthropicMessage>) -> Vec<AnthropicMessage>,
messages: Value,
) -> Value {
let parsed: Vec<AnthropicMessage> = serde_json::from_value(messages).unwrap();
fn apply(sanitizer: fn(Vec<Message>) -> Vec<Message>, messages: Value) -> Value {
let parsed: Vec<Message> = serde_json::from_value(messages).unwrap();
serde_json::to_value(sanitizer(parsed)).unwrap()
}
@ -631,11 +616,11 @@ mod tests {
serde_json::from_value(value).unwrap()
}
fn history(messages: Value) -> Vec<AnthropicMessage> {
fn history(messages: Value) -> Vec<Message> {
serde_json::from_value(messages).unwrap()
}
fn tools(value: Option<Value>) -> Option<Vec<Recognized<AnthropicTool>>> {
fn tools(value: Option<Value>) -> Option<Vec<Recognized<MessagesTool>>> {
value.map(|tools| serde_json::from_value(tools).unwrap())
}

View file

@ -1,4 +1,4 @@
use litellm_types::llms::anthropic_messages::anthropic_request::{AnthropicMessage, SystemPrompt};
use litellm_llms_types::formats::messages::{Message, SystemPrompt};
use serde::{Deserialize, Serialize};
use serde_json::Value;
@ -10,7 +10,7 @@ const TOKEN_COUNTING_BETA: &str = "token-counting-2024-11-01";
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
pub struct AnthropicCountTokensRequest {
pub model: String,
pub messages: Vec<AnthropicMessage>,
pub messages: Vec<Message>,
#[serde(skip_serializing_if = "Option::is_none")]
pub tools: Option<Vec<Value>>,
#[serde(skip_serializing_if = "Option::is_none")]
@ -25,12 +25,12 @@ pub struct AnthropicCountTokensResponse {
pub trait AnthropicCountTokensConfig {
fn endpoint(&self) -> &'static str;
fn validate_request(&self, model: &str, messages: &[AnthropicMessage]) -> Result<(), Error>;
fn validate_request(&self, model: &str, messages: &[Message]) -> Result<(), Error>;
fn transform_request(
&self,
model: &str,
messages: Vec<AnthropicMessage>,
messages: Vec<Message>,
tools: Option<Vec<Value>>,
system: Option<SystemPrompt>,
) -> Result<AnthropicCountTokensRequest, Error>;
@ -51,7 +51,7 @@ impl AnthropicCountTokensConfig for AnthropicCountTokensTransformation {
fn transform_request(
&self,
model: &str,
messages: Vec<AnthropicMessage>,
messages: Vec<Message>,
tools: Option<Vec<Value>>,
system: Option<SystemPrompt>,
) -> Result<AnthropicCountTokensRequest, Error> {
@ -65,7 +65,7 @@ impl AnthropicCountTokensConfig for AnthropicCountTokensTransformation {
})
}
fn validate_request(&self, model: &str, messages: &[AnthropicMessage]) -> Result<(), Error> {
fn validate_request(&self, model: &str, messages: &[Message]) -> Result<(), Error> {
if model.is_empty() {
return Err(Error::MissingField("model"));
}
@ -92,13 +92,13 @@ impl AnthropicCountTokensConfig for AnthropicCountTokensTransformation {
#[cfg(test)]
mod tests {
use litellm_types::llms::anthropic_messages::anthropic_request::MessageContent;
use litellm_llms_types::formats::messages::MessageContent;
use serde_json::{Map, json};
use super::*;
fn message() -> AnthropicMessage {
AnthropicMessage {
fn message() -> Message {
Message {
role: "user".into(),
content: MessageContent::Text("hello".into()),
extra: Map::new(),

View file

@ -1,9 +1,9 @@
This directory owns Anthropic's implementation of the Messages adapter contract in `base_llm/messages`. Shared Messages API data contracts belong in `litellm-types::messages`, and call orchestration belongs in `core/src/messages`. Sharing the `llms` crate with `base_llm/messages` does not erase this boundary
This directory owns Anthropic's implementation of the Messages adapter contract in `base_llm/messages`. Shared Messages API data contracts belong in `litellm-llms-types::formats::messages`, and call orchestration belongs in `core/src/messages`. Sharing the `llms` crate with `base_llm/messages` does not erase this boundary
Payload shaping, metadata filtering, tool-ID rewriting, web-search replay handling, thinking translation, and beta selection are provider policy. Keep them here or in Anthropic helpers shared by its operations. Pure payload shaping belongs with transformations, even if an existing file is named `handler.rs`
Bedrock and Azure adapters may explicitly reuse these helpers where Anthropic policy applies to their Claude backend. That reuse does not make the policy part of the shared Messages contract or a default for every provider. Shared `base_llm` code must never depend on this implementation
`web_search_result`, `web_search_tool_result_error`, and encrypted-content fields are protocol data owned by `litellm-types`. Keep those schemas separate from decisions about flattening, encrypted results, beta requirements, and model capabilities
`web_search_result`, `web_search_tool_result_error`, and encrypted-content fields are protocol data owned by `litellm-llms-types`. Keep those schemas separate from decisions about flattening, encrypted results, beta requirements, and model capabilities
Protocol reference: [Messages API](https://platform.claude.com/docs/en/api/http/messages/create)

View file

@ -1,7 +1,7 @@
use litellm_types::{
llms::anthropic_messages::anthropic_request::{
AdaptiveThinking, AnthropicMessage, AnthropicMessagesOptionalParams,
AnthropicMessagesRequest, EnabledThinking, ThinkingConfig, ThinkingDisplay,
use litellm_llms_types::{
formats::messages::{
AdaptiveThinking, EnabledThinking, Message, MessagesOptionalParams, MessagesRequest,
ThinkingConfig, ThinkingDisplay,
},
recognized::Recognized,
};
@ -16,12 +16,12 @@ use crate::{
};
pub fn shape_anthropic_messages_request(
request: AnthropicMessagesRequest,
request: MessagesRequest,
reasoning_auto_summary: bool,
) -> Result<AnthropicMessagesRequest, Error> {
Ok(AnthropicMessagesRequest {
) -> Result<MessagesRequest, Error> {
Ok(MessagesRequest {
messages: sanitize_anthropic_messages(request.messages),
params: AnthropicMessagesOptionalParams {
params: MessagesOptionalParams {
metadata: request
.params
.metadata
@ -35,7 +35,7 @@ pub fn shape_anthropic_messages_request(
})
}
fn sanitize_anthropic_messages(messages: Vec<AnthropicMessage>) -> Vec<AnthropicMessage> {
fn sanitize_anthropic_messages(messages: Vec<Message>) -> Vec<Message> {
strip_provider_specific_fields(flatten_unencrypted_web_search_results(
sanitize_tool_use_ids(strip_empty_content_blocks(messages)),
))
@ -100,11 +100,11 @@ mod tests {
use super::*;
fn messages(value: Value) -> Vec<AnthropicMessage> {
fn messages(value: Value) -> Vec<Message> {
serde_json::from_value(value).unwrap()
}
fn request(body: Value) -> AnthropicMessagesRequest {
fn request(body: Value) -> MessagesRequest {
serde_json::from_value(body).unwrap()
}

View file

@ -1,14 +1,14 @@
use litellm_python_compat::{json::from_json, repr::repr, truthy::truthy};
use litellm_types::{
llms::{
anthropic_messages::anthropic_request::{
AnthropicMessagesOptionalParams, AnthropicMessagesRequest, EffortLevel, OutputConfig,
ThinkingConfig, ThinkingDisplay,
use litellm_llms_types::{
formats::{
chat_completions::ReasoningEffort,
messages::{
EffortLevel, MessagesOptionalParams, MessagesRequest, OutputConfig, ThinkingConfig,
ThinkingDisplay,
},
openai::ReasoningEffort,
},
recognized::Recognized,
};
use litellm_python_compat::{json::from_json, repr::repr, truthy::truthy};
use serde_json::Value;
use crate::base_llm::messages::context::{
@ -84,11 +84,11 @@ fn fit_budget_to_max_tokens(budget_tokens: u64, max_tokens: Option<u64>) -> Opti
(max_tokens > ANTHROPIC_MIN_THINKING_BUDGET_TOKENS).then(|| budget_tokens.min(max_tokens - 1))
}
fn known_thinking(request: &AnthropicMessagesRequest) -> Option<&ThinkingConfig> {
fn known_thinking(request: &MessagesRequest) -> Option<&ThinkingConfig> {
request.params.thinking.as_ref().and_then(Recognized::known)
}
fn known_effort(request: &AnthropicMessagesRequest) -> Option<&Recognized<EffortLevel>> {
fn known_effort(request: &MessagesRequest) -> Option<&Recognized<EffortLevel>> {
request
.params
.output_config
@ -141,14 +141,14 @@ fn legacy_reasoning_effort(
}
fn translate_reasoning_effort(
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &ThinkingContext,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
let Some(reasoning_effort) = request.params.reasoning_effort else {
return Ok(request);
};
let request = AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
let request = MessagesRequest {
params: MessagesOptionalParams {
reasoning_effort: None,
..request.params
},
@ -165,8 +165,8 @@ fn translate_reasoning_effort(
output_effort(effort),
budget_for_effort(&context.budgets, effort),
) else {
return Ok(AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
return Ok(MessagesRequest {
params: MessagesOptionalParams {
thinking: None,
output_config: None,
..request.params
@ -180,8 +180,8 @@ fn translate_reasoning_effort(
return Err(unsupported_effort(level, &request.model));
}
let adaptive = ThinkingConfig::adaptive(Some(ThinkingDisplay::Summarized));
return Ok(AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
return Ok(MessagesRequest {
params: MessagesOptionalParams {
thinking: Some(
request
.params
@ -198,8 +198,8 @@ fn translate_reasoning_effort(
return Ok(request);
};
let enabled = ThinkingConfig::enabled(budget);
Ok(AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
Ok(MessagesRequest {
params: MessagesOptionalParams {
thinking: Some(
request
.params
@ -212,17 +212,14 @@ fn translate_reasoning_effort(
})
}
fn drop_disabled_thinking(
request: AnthropicMessagesRequest,
context: &ThinkingContext,
) -> AnthropicMessagesRequest {
fn drop_disabled_thinking(request: MessagesRequest, context: &ThinkingContext) -> MessagesRequest {
if !context.capabilities.thinking_always_on
|| !matches!(known_thinking(&request), Some(ThinkingConfig::Disabled(_)))
{
return request;
}
AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
MessagesRequest {
params: MessagesOptionalParams {
thinking: None,
..request.params
},
@ -231,9 +228,9 @@ fn drop_disabled_thinking(
}
fn translate_legacy_thinking_for_adaptive_model(
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &ThinkingContext,
) -> AnthropicMessagesRequest {
) -> MessagesRequest {
let capabilities = &context.capabilities;
if !capabilities.supports_adaptive_thinking || capabilities.supports_legacy_thinking {
return request;
@ -248,8 +245,8 @@ fn translate_legacy_thinking_for_adaptive_model(
.copied()
.unwrap_or(0);
let level = effort_for_budget(&context.budgets, budget, capabilities);
AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
MessagesRequest {
params: MessagesOptionalParams {
thinking: Some(Recognized::Known(ThinkingConfig::adaptive(None))),
output_config: with_default_effort(request.params.output_config, level),
..request.params
@ -259,9 +256,9 @@ fn translate_legacy_thinking_for_adaptive_model(
}
fn translate_adaptive_effort_for_non_adaptive_model(
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &ThinkingContext,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
let capabilities = &context.capabilities;
if capabilities.supports_adaptive_thinking {
return Ok(request);
@ -276,8 +273,8 @@ fn translate_adaptive_effort_for_non_adaptive_model(
_ => true,
};
if supports_effort_param(capabilities) && (!adaptive_thinking || level_accepted) {
return Ok(AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
return Ok(MessagesRequest {
params: MessagesOptionalParams {
thinking: if adaptive_thinking {
None
} else {
@ -293,8 +290,8 @@ fn translate_adaptive_effort_for_non_adaptive_model(
} else {
None
};
Ok(AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
Ok(MessagesRequest {
params: MessagesOptionalParams {
thinking: budget
.and_then(|budget| fit_budget_to_max_tokens(budget, request.params.max_tokens))
.map(|budget| Recognized::Known(ThinkingConfig::enabled(budget))),
@ -306,9 +303,9 @@ fn translate_adaptive_effort_for_non_adaptive_model(
}
fn drop_incompatible_temperature_for_thinking(
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &ThinkingContext,
) -> AnthropicMessagesRequest {
) -> MessagesRequest {
if context.capabilities.supports_adaptive_thinking {
return request;
}
@ -321,8 +318,8 @@ fn drop_incompatible_temperature_for_thinking(
if !pinned || !(thinking_enabled || effort_enabled) {
return request;
}
AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
MessagesRequest {
params: MessagesOptionalParams {
temperature: None,
..request.params
},
@ -331,9 +328,9 @@ fn drop_incompatible_temperature_for_thinking(
}
pub fn translate_thinking(
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &ThinkingContext,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
let request = translate_reasoning_effort(request, context)?;
let request = drop_disabled_thinking(request, context);
let request = translate_legacy_thinking_for_adaptive_model(request, context);
@ -350,7 +347,7 @@ mod tests {
const EFFORT_CHOICES: &str = "'none', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max'";
fn request(fields: Value) -> AnthropicMessagesRequest {
fn request(fields: Value) -> MessagesRequest {
let mut body = serde_json::json!({"model": "claude", "messages": [{"role": "user", "content": "Hello"}]});
body.as_object_mut()
.unwrap()
@ -368,7 +365,7 @@ mod tests {
fn translate(
capabilities: MessagesModelCapabilities,
fields: Value,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
translate_thinking(request(fields), &context(capabilities))
}

View file

@ -1,12 +1,9 @@
use litellm_auth::CredentialPlacement;
use litellm_types::{
llms::{
anthropic::{AnthropicBeta, BetaSet},
anthropic_messages::anthropic_request::{
AnthropicMessage, AnthropicMessagesOptionalParams, AnthropicMessagesRequest,
ContextEdit, ContextManagement, Speed,
},
use litellm_llms_types::{
formats::messages::{
ContextEdit, ContextManagement, Message, MessagesOptionalParams, MessagesRequest, Speed,
},
providers::anthropic::{AnthropicBeta, BetaSet},
recognized::Recognized,
};
use serde_json::{Map, Value, json};
@ -24,7 +21,7 @@ use crate::{
},
base_llm::{
auth::AuthScheme,
messages::transformation::{BaseAnthropicMessagesConfig, Headers, ValidatedEnvironment},
messages::transformation::{BaseMessagesConfig, Headers, ValidatedEnvironment},
},
};
@ -37,12 +34,12 @@ pub struct AnthropicMessagesConfig;
pub const ANTHROPIC_MESSAGES_CONFIG: AnthropicMessagesConfig = AnthropicMessagesConfig;
impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig {
impl BaseMessagesConfig for AnthropicMessagesConfig {
fn shape_request(
&self,
request: AnthropicMessagesRequest,
request: MessagesRequest,
reasoning_auto_summary: bool,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
shape_anthropic_messages_request(request, reasoning_auto_summary)
}
@ -57,9 +54,9 @@ impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig {
fn transform_anthropic_messages_request(
&self,
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &MessagesTransformContext,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
transform_messages_request(request, context)
}
@ -112,15 +109,15 @@ impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig {
DEFAULT_HEADERS
}
fn request_headers(&self, headers: Headers, request: &AnthropicMessagesRequest) -> Headers {
fn request_headers(&self, headers: Headers, request: &MessagesRequest) -> Headers {
update_headers_with_anthropic_beta(headers, request)
}
}
pub(crate) fn transform_messages_request(
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &MessagesTransformContext,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
if request.params.max_tokens.is_none() {
return Err(Error::MissingField("max_tokens"));
}
@ -136,9 +133,9 @@ pub(crate) fn transform_messages_request(
} else {
strip_advisor_blocks(request.messages)
};
Ok(AnthropicMessagesRequest {
Ok(MessagesRequest {
messages: strip_encrypted_reasoning_blocks(messages),
params: AnthropicMessagesOptionalParams {
params: MessagesOptionalParams {
context_management,
..request.params
},
@ -148,12 +145,12 @@ pub(crate) fn transform_messages_request(
pub(crate) fn update_headers_with_anthropic_beta(
headers: Headers,
request: &AnthropicMessagesRequest,
request: &MessagesRequest,
) -> Headers {
merge_beta_headers(headers, feature_betas(request))
}
fn feature_betas(request: &AnthropicMessagesRequest) -> BetaSet {
fn feature_betas(request: &MessagesRequest) -> BetaSet {
let params = &request.params;
let tools = params.tools.as_deref();
[
@ -192,7 +189,7 @@ fn context_management_betas(
.chain(other.then_some(AnthropicBeta::ContextManagement20250627))
}
fn uses_structured_output(params: &AnthropicMessagesOptionalParams) -> bool {
fn uses_structured_output(params: &MessagesOptionalParams) -> bool {
params.output_format.is_some()
|| params
.output_config
@ -201,7 +198,7 @@ fn uses_structured_output(params: &AnthropicMessagesOptionalParams) -> bool {
.is_some_and(|config| config.format.is_some())
}
fn messages_carry_output_config(messages: &[AnthropicMessage]) -> bool {
fn messages_carry_output_config(messages: &[Message]) -> bool {
messages
.iter()
.any(|message| message.extra.contains_key("output_config"))
@ -217,9 +214,9 @@ fn unsupported_param(model: &str, param: &str, value: &str, hint: &str) -> Error
}
fn drop_unsupported_params(
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &MessagesTransformContext,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
let capabilities = &context.thinking.capabilities;
let model = request.model.clone();
let reject = |param: &str, value: String, hint: &str| -> Result<(), Error> {
@ -237,8 +234,8 @@ fn drop_unsupported_params(
_ => params.speed.clone(),
};
if capabilities.supports_sampling_params {
return Ok(AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams { speed, ..params },
return Ok(MessagesRequest {
params: MessagesOptionalParams { speed, ..params },
..request
});
}
@ -259,8 +256,8 @@ fn drop_unsupported_params(
if let Some(top_k) = params.top_k {
reject("top_k", json!(top_k).to_string(), "")?;
}
Ok(AnthropicMessagesRequest {
params: AnthropicMessagesOptionalParams {
Ok(MessagesRequest {
params: MessagesOptionalParams {
speed,
temperature,
top_p: None,
@ -366,7 +363,7 @@ mod tests {
)
}
fn request(fields: Value) -> AnthropicMessagesRequest {
fn request(fields: Value) -> MessagesRequest {
serde_json::from_value(body(fields)).unwrap()
}

View file

@ -12,10 +12,10 @@ use crate::base_llm::ocr::{
error::Error,
handler::OcrClient,
transformation::{
BaseOcrConfig, LiteLLMOcrResponse, OcrDocument, OcrRequestContext, OcrResponseFormat,
PreparedOcrRequest, decode_and_normalize_response,
BaseOcrConfig, OcrRequestContext, PreparedOcrRequest, decode_and_normalize_response,
},
};
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument, OcrResponseFormat};
const DEFAULT_FEATURE_TYPES: [FeatureType; 2] = [FeatureType::Layout, FeatureType::Tables];

View file

@ -7,11 +7,9 @@ use strum::{EnumString, IntoStaticStr, VariantNames};
use crate::base_llm::ocr::{
document::{InlineDocument, inline_remote_document},
error::Error,
transformation::{
LiteLLMOcrResponse, OcrDocument, OcrEnvironment, OcrPage, OcrRequestContext, OcrUsageInfo,
PreparedOcrRequest,
},
transformation::{OcrEnvironment, OcrRequestContext, PreparedOcrRequest},
};
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument, OcrPage, OcrUsageInfo};
const TEXTRACT_SERVICE: &str = "textract";
const AWS_JSON_CONTENT_TYPE: &str = "application/x-amz-json-1.1";

View file

@ -10,10 +10,10 @@ use crate::base_llm::ocr::{
error::Error,
handler::OcrClient,
transformation::{
BaseOcrConfig, LiteLLMOcrResponse, OcrDocument, OcrRequestContext, OcrResponseFormat,
PreparedOcrRequest, decode_and_normalize_response,
BaseOcrConfig, OcrRequestContext, PreparedOcrRequest, decode_and_normalize_response,
},
};
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument, OcrResponseFormat};
#[derive(Debug, Deserialize, Serialize)]
pub struct DetectDocumentTextRequest {

View file

@ -1,3 +1,3 @@
This directory owns Azure's Messages adapter: its endpoints, authentication policy, headers, and transformations. Implement the shared adapter contract from `base_llm/messages`, consume API data contracts from `litellm-types::messages`, and leave call orchestration to `core/src/messages`
This directory owns Azure's Messages adapter: its endpoints, authentication policy, headers, and transformations. Implement the shared adapter contract from `base_llm/messages`, consume API data contracts from `litellm-llms-types::formats::messages`, and leave call orchestration to `core/src/messages`
The Claude adapter may explicitly reuse payload policy from `anthropic/messages` when it applies to Azure's Claude backend. Keep Azure-specific differences here. Sharing that helper does not make Anthropic policy a format-wide default or justify a dependency from `base_llm/messages` on provider implementations

View file

@ -1,8 +1,8 @@
use litellm_auth::{CredentialPlacement, SecretValue};
use litellm_http::request::{has_bearer_auth, has_header};
use litellm_types::llms::anthropic_messages::anthropic_request::{
AnthropicMessage, AnthropicMessagesOptionalParams, AnthropicMessagesRequest, CacheControl,
ContentBlock, MessageContent, SystemPrompt,
use litellm_llms_types::formats::messages::{
CacheControl, ContentBlock, Message, MessageContent, MessagesOptionalParams, MessagesRequest,
SystemPrompt,
};
use crate::{
@ -21,7 +21,7 @@ use crate::{
messages::{
context::MessagesTransformContext,
normalization::fold_system_role_messages,
transformation::{BaseAnthropicMessagesConfig, MESSAGES_PATH_SUFFIX},
transformation::{BaseMessagesConfig, MESSAGES_PATH_SUFFIX},
},
},
};
@ -34,12 +34,12 @@ pub struct AzureAnthropicMessagesConfig;
pub const AZURE_ANTHROPIC_MESSAGES_CONFIG: AzureAnthropicMessagesConfig =
AzureAnthropicMessagesConfig;
impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig {
impl BaseMessagesConfig for AzureAnthropicMessagesConfig {
fn shape_request(
&self,
request: AnthropicMessagesRequest,
request: MessagesRequest,
reasoning_auto_summary: bool,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
shape_anthropic_messages_request(request, reasoning_auto_summary)
}
@ -54,18 +54,18 @@ impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig {
fn transform_anthropic_messages_request(
&self,
request: AnthropicMessagesRequest,
request: MessagesRequest,
context: &MessagesTransformContext,
) -> Result<AnthropicMessagesRequest, Error> {
) -> Result<MessagesRequest, Error> {
let request = fold_system_role_messages(request);
transform_messages_request(
AnthropicMessagesRequest {
MessagesRequest {
messages: request
.messages
.into_iter()
.map(strip_scope_from_message)
.collect(),
params: AnthropicMessagesOptionalParams {
params: MessagesOptionalParams {
system: request.params.system.map(strip_scope_from_system),
..request.params
},
@ -105,7 +105,7 @@ impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig {
DEFAULT_HEADERS
}
fn request_headers(&self, headers: Headers, request: &AnthropicMessagesRequest) -> Headers {
fn request_headers(&self, headers: Headers, request: &MessagesRequest) -> Headers {
update_headers_with_anthropic_beta(headers, request)
}
}
@ -148,8 +148,8 @@ fn strip_scope_from_system(system: SystemPrompt) -> SystemPrompt {
}
}
fn strip_scope_from_message(message: AnthropicMessage) -> AnthropicMessage {
AnthropicMessage {
fn strip_scope_from_message(message: Message) -> Message {
Message {
content: match message.content {
MessageContent::Blocks(blocks) => {
MessageContent::Blocks(blocks.into_iter().map(strip_scope_from_block).collect())
@ -162,7 +162,7 @@ fn strip_scope_from_message(message: AnthropicMessage) -> AnthropicMessage {
#[cfg(test)]
mod tests {
use litellm_types::llms::anthropic_messages::anthropic_response::AnthropicMessagesResponse;
use litellm_llms_types::formats::messages::MessagesResponse;
use rstest::rstest;
use serde_json::json;
@ -171,11 +171,11 @@ mod tests {
use super::*;
use crate::base_llm::messages::context::MessagesModelCapabilities;
fn request_from(value: serde_json::Value) -> AnthropicMessagesRequest {
fn request_from(value: serde_json::Value) -> MessagesRequest {
serde_json::from_value(value).expect("valid request")
}
fn to_value(request: AnthropicMessagesRequest) -> serde_json::Value {
fn to_value(request: MessagesRequest) -> serde_json::Value {
serde_json::to_value(request).expect("serializable request")
}
@ -518,7 +518,7 @@ mod tests {
#[test]
fn transform_request_rejects_non_object_body() {
let err = serde_json::from_value::<AnthropicMessagesRequest>(json!("bad"))
let err = serde_json::from_value::<MessagesRequest>(json!("bad"))
.expect_err("non-object body should error");
assert!(err.is_data());
}
@ -576,7 +576,7 @@ mod tests {
#[test]
fn transform_response_passes_through() {
let response: AnthropicMessagesResponse = serde_json::from_value(json!({
let response: MessagesResponse = serde_json::from_value(json!({
"id": "msg_1",
"type": "message",
"role": "assistant",

View file

@ -6,15 +6,13 @@ use crate::{
document::{inline_remote_document, validate_inline_document},
error::Error,
handler::OcrClient,
transformation::{
BaseOcrConfig, LiteLLMOcrResponse, OcrDocument, OcrRequestContext, OcrResponseFormat,
PreparedOcrRequest,
},
transformation::{BaseOcrConfig, OcrRequestContext, PreparedOcrRequest},
},
cohere::ocr::transformation::{
CohereOptions, CohereParseConfig, CohereRequest, validate_document,
},
};
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument, OcrResponseFormat};
pub const AZURE_COHERE_PARSE_PATH: [&str; 4] = ["providers", "cohere", "v2", "parse"];

View file

@ -3,11 +3,8 @@ use std::{collections::BTreeSet, time::Duration};
use base64::{Engine, engine::general_purpose::STANDARD};
use litellm_auth::{InputSource, Sourced};
use litellm_auth_azure::{AzureAuthInputs, SECRET_NAMES as AZURE_AUTH_SECRET_NAMES};
use litellm_core_utils::{
call_arguments::CallArguments,
serde_compat::{FiniteF64, LaxI64},
url_utils::ApiUrl,
};
use litellm_core_utils::{call_arguments::CallArguments, url_utils::ApiUrl};
use litellm_llms_types::serde_compat::{FiniteF64, LaxI64};
use reqwest::Url;
use serde::{Deserialize, Deserializer, Serialize};
use serde_json::{Map, Value};
@ -20,12 +17,14 @@ use crate::base_llm::ocr::{
handler::{CallHooks, OcrClient, read_json_response},
settings::OcrSettings,
transformation::{
BaseOcrConfig, DecodedOcrResponse, LiteLLMOcrResponse, OCR_INLINE_MAX_BYTES,
OCR_POLL_RETRY_SECS, OcrConnection, OcrCredentialInputs, OcrDocument, OcrPage,
OcrPageDimensions, OcrResponseContext, OcrResponseFormat, OcrUsageInfo, PreparedOcrRequest,
BaseOcrConfig, DecodedOcrResponse, OCR_INLINE_MAX_BYTES, OCR_POLL_RETRY_SECS,
OcrConnection, OcrCredentialInputs, OcrResponseContext, PreparedOcrRequest,
ResolvedOcrCredentials, decode_and_normalize_response, decode_response,
},
};
use litellm_llms_types::formats::ocr::{
LiteLLMOcrResponse, OcrDocument, OcrPage, OcrPageDimensions, OcrResponseFormat, OcrUsageInfo,
};
const AZURE_DI_SUBSCRIPTION_HEADER: &str = "Ocp-Apim-Subscription-Key";
const AZURE_DI_DEFAULT_WIDTH: f64 = 8.5;

View file

@ -1,6 +1,5 @@
use litellm_auth::{InputSource, Sourced};
use litellm_auth_azure::AzureAuthInputs;
use litellm_auth_azure::SECRET_NAMES as AZURE_AUTH_SECRET_NAMES;
use litellm_auth_azure::{AzureAuthInputs, SECRET_NAMES as AZURE_AUTH_SECRET_NAMES};
use litellm_core_utils::{call_arguments::CallArguments, params::OpaqueParams, url_utils::ApiUrl};
use serde_json::Value;
@ -9,13 +8,11 @@ use crate::{
document::{inline_remote_document, validate_inline_document},
error::Error,
handler::OcrClient,
transformation::{
BaseOcrConfig, LiteLLMOcrResponse, OcrConnection, OcrDocument, OcrRequestContext,
OcrResponseFormat, PreparedOcrRequest,
},
transformation::{BaseOcrConfig, OcrConnection, OcrRequestContext, PreparedOcrRequest},
},
mistral::ocr::transformation::{MistralOcrConfig, MistralOcrRequest},
};
use litellm_llms_types::formats::ocr::{LiteLLMOcrResponse, OcrDocument, OcrResponseFormat};
pub const AZURE_AI_OCR_PATH: [&str; 4] = ["providers", "mistral", "azure", "ocr"];

View file

@ -1,4 +1,4 @@
use litellm_types::audio_transcription::AudioTranscriptionResponseData;
use litellm_llms_types::formats::audio_transcription::AudioTranscriptionResponseData;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};

View file

@ -1,7 +1,7 @@
use std::collections::HashMap;
use futures_util::{StreamExt, stream::BoxStream};
use litellm_types::utils::ChatCompletionChunk;
use litellm_llms_types::formats::chat_completions::ChatCompletionChunk;
use crate::{
Error,

View file

@ -1,6 +1,5 @@
use litellm_types::{
llms::openai::{ChatMessage, ChatMessageContent},
utils::ChatCompletionsResponse,
use litellm_llms_types::formats::chat_completions::{
ChatCompletionsResponse, ChatMessage, ChatMessageContent,
};
use serde_json::{Map, Value};

View file

@ -1,4 +1,4 @@
This directory owns the shared Messages provider adapter contract, its execution inputs such as `MessagesTransformContext`, and provider-independent transformation machinery. Public request, response, content-block, and event schemas belong in `litellm-types::messages`. Call orchestration belongs in `core/src/messages`, and provider implementations belong in `llms/src/<provider>/messages`
This directory owns the shared Messages provider adapter contract, its execution inputs such as `MessagesTransformContext`, and provider-independent transformation machinery. Public request, response, content-block, and event schemas belong in `litellm-llms-types::formats::messages`. Call orchestration belongs in `core/src/messages`, and provider implementations belong in `llms/src/<provider>/messages`
Do not import provider implementations or embed their policy in shared trait defaults, normalization, or context defaults. A context carries inputs the shared adapter contract needs, not every provider's settings. Thinking-budget choices and model-specific restrictions do not become format rules merely because several providers host Claude

View file

@ -1,6 +1,5 @@
use litellm_types::llms::anthropic_messages::anthropic_request::{
AnthropicMessage, AnthropicMessagesOptionalParams, AnthropicMessagesRequest, ContentBlock,
MessageContent, SystemPrompt,
use litellm_llms_types::formats::messages::{
ContentBlock, Message, MessageContent, MessagesOptionalParams, MessagesRequest, SystemPrompt,
};
const SYSTEM_ROLE: &str = "system";
@ -20,12 +19,12 @@ fn system_into_blocks(system: Option<SystemPrompt>) -> Vec<ContentBlock> {
}
}
pub fn fold_system_role_messages(request: AnthropicMessagesRequest) -> AnthropicMessagesRequest {
pub fn fold_system_role_messages(request: MessagesRequest) -> MessagesRequest {
if !request.messages.iter().any(|msg| msg.role == SYSTEM_ROLE) {
return request;
}
let (system_messages, chat_messages): (Vec<AnthropicMessage>, Vec<AnthropicMessage>) = request
let (system_messages, chat_messages): (Vec<Message>, Vec<Message>) = request
.messages
.into_iter()
.partition(|msg| msg.role == SYSTEM_ROLE);
@ -39,9 +38,9 @@ pub fn fold_system_role_messages(request: AnthropicMessagesRequest) -> Anthropic
)
.collect();
AnthropicMessagesRequest {
MessagesRequest {
messages: chat_messages,
params: AnthropicMessagesOptionalParams {
params: MessagesOptionalParams {
system: (!folded_system.is_empty()).then_some(SystemPrompt::Blocks(folded_system)),
..request.params
},

View file

@ -1,7 +1,7 @@
use bytes::Bytes;
use futures_util::{StreamExt, stream::BoxStream};
use litellm_framing::{frames, sse::SseCodec};
use litellm_types::messages::streaming::MessagesStreamEvent;
use litellm_llms_types::formats::messages::streaming::MessagesStreamEvent;
use crate::Error;
pub use crate::base_llm::base_model_iterator::ByteStream;
@ -38,7 +38,9 @@ pub fn encode_anthropic_sse(event: &MessagesStreamEvent) -> Result<Bytes, Error>
#[cfg(test)]
mod tests {
use futures_util::{StreamExt, TryStreamExt, stream};
use litellm_types::messages::streaming::{MessagesContentBlockDelta, MessagesStreamUsage};
use litellm_llms_types::formats::messages::streaming::{
MessagesContentBlockDelta, MessagesStreamUsage,
};
use serde_json::json;
use super::*;

Some files were not shown because too many files have changed in this diff Show more