merge: resolve conflict with main in member budget seeding

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Devin AI 2026-09-17 23:42:41 +00:00
commit 569ef3ea25
96 changed files with 1431 additions and 316 deletions

View file

@ -96,6 +96,7 @@ GATEWAY_PATH_PREFIXES: tuple[str, ...] = (
"/langfuse/",
"/vllm/",
"/mistral/",
"/typesafe/",
"/nvidia_nim/",
"/groq/",
"/voyage/",

View file

@ -2016,6 +2016,7 @@ dependencies = [
"litellm-auth-azure",
"litellm-auth-gcp",
"litellm-framing",
"litellm-providers",
"mime_guess",
"moka",
"rand 0.8.7",
@ -2052,6 +2053,18 @@ dependencies = [
"tokio",
]
[[package]]
name = "litellm-providers"
version = "0.1.0"
dependencies = [
"litellm-auth",
"litellm-auth-aws",
"rstest",
"serde",
"serde_json",
"thiserror 2.0.19",
]
[[package]]
name = "litellm-python-bridge"
version = "0.1.0"

View file

@ -16,6 +16,7 @@ litellm-auth = { path = "crates/auth" }
litellm-auth-aws = { path = "crates/auth-aws" }
litellm-auth-azure = { path = "crates/auth-azure" }
litellm-auth-gcp = { path = "crates/auth-gcp" }
litellm-providers = { path = "crates/providers" }
litellm-cache = { path = "crates/cache" }
litellm-cache-memory = { path = "crates/cache-memory" }
litellm-token-counter = { path = "crates/token-counter" }

View file

@ -15,6 +15,7 @@ litellm-auth.workspace = true
litellm-auth-aws.workspace = true
litellm-auth-azure.workspace = true
litellm-auth-gcp.workspace = true
litellm-providers.workspace = true
litellm-framing.workspace = true
moka.workspace = true
mime_guess = "2.0.5"

View file

@ -24,3 +24,23 @@ pub enum Error {
#[error(transparent)]
Aws(#[from] litellm_auth_aws::Error),
}
impl From<litellm_providers::audio_transcription::Error> for Error {
fn from(error: litellm_providers::audio_transcription::Error) -> Self {
match error {
litellm_providers::audio_transcription::Error::InvalidType { expected, actual } => {
Self::InvalidType { expected, actual }
}
litellm_providers::audio_transcription::Error::MissingField(field) => {
Self::MissingField(field)
}
litellm_providers::audio_transcription::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::audio_transcription::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::audio_transcription::Error::Auth(error) => Self::Auth(error),
}
}
}

View file

@ -47,8 +47,8 @@ async fn signed_headers(
use std::collections::BTreeMap;
use std::time::SystemTime;
use crate::llms::base_llm::audio_transcription::transformation::AudioTranscriptionAuth;
use litellm_auth_aws::{aws_auth_config, resolve_credentials, sign_bedrock_post};
use litellm_providers::base_llm::audio_transcription::transformation::AudioTranscriptionAuth;
let AudioTranscriptionAuth::AwsSigV4 { region, .. } = &request.auth else {
return Ok(request.upstream_headers.clone());

View file

@ -3,7 +3,7 @@ pub use error::Error;
mod client;
mod handler;
mod prepare;
pub mod types;
pub use litellm_providers::audio_transcription::types;
pub use handler::execute_audio_transcription_provider_call;
pub use prepare::prepare_audio_transcription_provider_call;

View file

@ -4,10 +4,10 @@ use crate::http_utils::{has_header, string_headers};
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::audio_transcription::transformation::{
use litellm_providers::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
use crate::llms::bedrock::audio_transcription::BEDROCK_AUDIO_TRANSCRIPTION_CONFIG;
use litellm_providers::bedrock::audio_transcription::BEDROCK_AUDIO_TRANSCRIPTION_CONFIG;
fn provider_config(provider: &str) -> Option<&'static dyn BaseAudioTranscriptionConfig> {
if provider == "bedrock" {

View file

@ -2,8 +2,8 @@ use serde_json::{Map, Value};
use super::Error;
use crate::http_utils::string_headers as shared_string_headers;
use crate::llms::anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG;
use crate::llms::base_llm::chat::transformation::BaseConfig;
use litellm_providers::anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG;
use litellm_providers::base_llm::chat::transformation::BaseConfig;
const HEADER_CONTEXT: &str = "chat completions";
@ -11,7 +11,7 @@ pub(super) fn chat_completions_provider_config(provider: &str) -> Option<&'stati
match provider {
"anthropic" => Some(&ANTHROPIC_CHAT_COMPLETIONS_CONFIG),
"bedrock" => Some(
&crate::llms::bedrock::chat::converse_transformation::BEDROCK_CHAT_COMPLETIONS_CONFIG,
&litellm_providers::bedrock::chat::converse_transformation::BEDROCK_CHAT_COMPLETIONS_CONFIG,
),
_ => None,
}

View file

@ -24,3 +24,19 @@ pub enum Error {
#[error(transparent)]
Aws(#[from] litellm_auth_aws::Error),
}
impl From<litellm_providers::chat::Error> for Error {
fn from(error: litellm_providers::chat::Error) -> Self {
match error {
litellm_providers::chat::Error::MissingField(field) => Self::MissingField(field),
litellm_providers::chat::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::chat::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::chat::Error::Unsupported(reason) => Self::Unsupported(reason),
litellm_providers::chat::Error::Auth(error) => Self::Auth(error),
}
}
}

View file

@ -8,7 +8,7 @@ use super::types::{
ResolvedChatCompletionsRequest,
};
use crate::http_utils::{http_request, truncate_error_body};
use crate::llms::base_llm::chat::transformation::ChatCompletionsAuth;
use litellm_providers::base_llm::chat::transformation::ChatCompletionsAuth;
pub(super) async fn execute_chat_completions_provider_call(
request: ResolvedChatCompletionsRequest<'_>,
@ -59,6 +59,7 @@ pub(super) async fn execute_chat_completions_provider_call(
request
.config
.transform_response(&request.model, ProviderChatResponseData { body })
.map_err(Error::from)
.map_err(as_response_error)
}

View file

@ -10,12 +10,11 @@ mod error;
pub use error::Error;
mod client;
mod common_utils;
pub mod conversation;
pub use litellm_providers::chat::{conversation, response_utils};
pub(crate) mod handler;
mod prepare;
pub mod response_utils;
pub mod streaming;
pub mod types;
pub use litellm_providers::chat::types;
use handler::execute_chat_completions_provider_call;
use prepare::{parse_messages, resolve_provider_config, resolve_request};

View file

@ -10,7 +10,7 @@ use crate::http_utils::has_header;
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
use litellm_providers::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
pub(super) fn resolve_provider_config<'a>(
model: &'a str,

View file

@ -3,7 +3,7 @@ use serde_json::{Map, Value, json};
use super::Error;
use super::prepare::{prepare_provider_request, resolve_request};
use super::types::{ChatCompletionsRequest, ProviderChatCompletionsRequest};
use crate::llms::base_llm::chat::transformation::ChatCompletionsAuth;
use litellm_providers::base_llm::chat::transformation::ChatCompletionsAuth;
fn prepare_chat_completions_call(
request: ChatCompletionsRequest<'_>,

View file

@ -1,36 +1,4 @@
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct CustomLlmProvider<'a> {
pub model: &'a str,
pub custom_llm_provider: &'a str,
}
pub fn get_custom_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
) -> Option<CustomLlmProvider<'a>> {
if let Some(custom_llm_provider) = custom_llm_provider.filter(|provider| !provider.is_empty()) {
return Some(CustomLlmProvider {
model: strip_custom_llm_provider_prefix(model, custom_llm_provider),
custom_llm_provider,
});
}
let (custom_llm_provider, model) = model.split_once('/')?;
if custom_llm_provider.is_empty() || model.is_empty() {
return None;
}
Some(CustomLlmProvider {
model,
custom_llm_provider,
})
}
fn strip_custom_llm_provider_prefix<'a>(model: &'a str, custom_llm_provider: &str) -> &'a str {
model
.strip_prefix(custom_llm_provider)
.and_then(|model| model.strip_prefix('/'))
.unwrap_or(model)
}
pub use litellm_providers::provider_resolution::{CustomLlmProvider, get_custom_llm_provider};
#[cfg(test)]
mod tests {

View file

@ -1,2 +1 @@
pub mod streaming;
pub mod transformation;

View file

@ -2,16 +2,16 @@ use std::collections::HashMap;
use serde_json::Value;
use super::super::experimental_pass_through::messages::streaming::{
AnthropicContentBlock, AnthropicContentBlockDelta, AnthropicMessagesStreamEvent,
AnthropicStreamUsage,
};
use crate::chat_completions::Error;
use crate::chat_completions::streaming::StreamTransformer;
use crate::chat_completions::types::{
ChatCompletionChunk, ChatCompletionThinkingBlock, ChatCompletionToolCallChunk,
ChatCompletionsUsage,
};
use crate::llms::anthropic::experimental_pass_through::messages::streaming::{
AnthropicContentBlock, AnthropicContentBlockDelta, AnthropicMessagesStreamEvent,
AnthropicStreamUsage,
};
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum AnthropicJsonChunkType {

View file

@ -3,9 +3,9 @@ use serde_json::Value;
use time::OffsetDateTime;
use url::Url;
use crate::llms::anthropic::experimental_pass_through::messages::transformation::resolve_anthropic_api_base;
use crate::messages::Error;
use crate::messages::types::AnthropicMessagesResponse;
use litellm_providers::anthropic::experimental_pass_through::messages::transformation::resolve_anthropic_api_base;
const BATCHES_PATH_SUFFIX: &str = "/v1/messages/batches";

View file

@ -1,4 +1,3 @@
pub mod batches;
pub mod count_tokens;
pub mod streaming;
pub mod transformation;

View file

@ -1,2 +1 @@
pub mod anthropic;
pub(crate) mod ocr;

View file

@ -1,4 +1 @@
pub mod anthropic_messages;
pub mod audio_transcription;
pub mod chat;
pub(crate) mod ocr;

View file

@ -1,7 +1,6 @@
pub mod anthropic;
pub mod azure_ai;
pub mod base_llm;
pub mod bedrock;
pub(crate) mod cohere;
pub(crate) mod mistral;
pub mod openai;

View file

@ -3,9 +3,9 @@ use serde_json::{Map, Value};
use super::Error;
use crate::http_utils::string_headers as shared_string_headers;
pub(super) use crate::http_utils::{has_bearer_auth, has_header, truncate_error_body};
use crate::llms::anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG;
use crate::llms::azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG;
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use litellm_providers::anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG;
use litellm_providers::azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG;
use litellm_providers::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
const HEADER_CONTEXT: &str = "messages";

View file

@ -28,6 +28,22 @@ pub enum Error {
InvalidBedrockBase64(String),
}
impl From<litellm_providers::messages::Error> for Error {
fn from(error: litellm_providers::messages::Error) -> Self {
match error {
litellm_providers::messages::Error::MissingField(field) => Self::MissingField(field),
litellm_providers::messages::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::messages::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::messages::Error::Unsupported(reason) => Self::Unsupported(reason),
litellm_providers::messages::Error::Auth(error) => Self::Auth(error),
}
}
}
impl Error {
pub fn is_request(&self) -> bool {
match self {

View file

@ -40,6 +40,7 @@ pub(super) async fn execute_messages_provider_call(
request
.config
.transform_anthropic_messages_response(&request.model, response)
.map_err(Error::from)
}
pub(super) async fn execute_messages_provider_stream(

View file

@ -13,7 +13,7 @@ mod client;
mod common_utils;
mod handler;
mod prepare;
pub mod types;
pub use litellm_providers::messages::types;
use handler::{execute_messages_provider_call, execute_messages_provider_stream};
use types::{AnthropicMessagesResponse, MessagesRequest};

View file

@ -6,7 +6,7 @@ use super::types::{MessagesRequest, ProviderMessagesRequest};
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::anthropic_messages::transformation::{
use litellm_providers::base_llm::anthropic_messages::transformation::{
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
};

View file

@ -0,0 +1,16 @@
[package]
name = "litellm-providers"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
litellm-auth.workspace = true
litellm-auth-aws.workspace = true
serde.workspace = true
serde_json.workspace = true
thiserror.workspace = true
[dev-dependencies]
rstest.workspace = true

View file

@ -1,7 +1,7 @@
use serde_json::json;
use super::*;
use crate::chat_completions::Error;
use crate::chat::Error;
fn messages(value: Value) -> Vec<ChatMessage> {
serde_json::from_value(value).expect("valid messages")

View file

@ -1,19 +1,19 @@
use serde_json::{Map, Value, json};
use crate::chat_completions::Error;
use crate::chat_completions::conversation::{Conversation, build_conversation};
use crate::chat_completions::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat_completions::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
ProviderChatRequestData, ProviderChatResponseData,
};
use crate::constants::ANTHROPIC_OAUTH_TOKEN_PREFIX;
use crate::llms::anthropic::experimental_pass_through::messages::transformation::{
use crate::anthropic::ANTHROPIC_OAUTH_TOKEN_PREFIX;
use crate::anthropic::experimental_pass_through::messages::transformation::{
complete_anthropic_url, resolve_anthropic_api_key,
};
use crate::llms::base_llm::chat::transformation::{
use crate::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use crate::chat::Error;
use crate::chat::conversation::{Conversation, build_conversation};
use crate::chat::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
ProviderChatRequestData, ProviderChatResponseData,
};
/// Anthropic parameter names, post `map_openai_params`, that the Rust path can
/// place verbatim in the Messages body.

View file

@ -1,4 +1,4 @@
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::messages::Error;
const ANTHROPIC_API_KEY_ENV: &str = "ANTHROPIC_API_KEY";

View file

@ -0,0 +1 @@
pub mod messages;

View file

@ -0,0 +1,4 @@
pub mod chat;
pub mod experimental_pass_through;
pub const ANTHROPIC_OAUTH_TOKEN_PREFIX: &str = "sk-ant-oat";

View file

@ -0,0 +1,31 @@
use thiserror::Error;
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("expected {expected}, got {actual}")]
InvalidType {
expected: &'static str,
actual: &'static str,
},
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub fn json_type_name(value: &serde_json::Value) -> &'static str {
match value {
serde_json::Value::Null => "null",
serde_json::Value::Bool(_) => "boolean",
serde_json::Value::Number(_) => "number",
serde_json::Value::String(_) => "string",
serde_json::Value::Array(_) => "array",
serde_json::Value::Object(_) => "object",
}
}
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::audio_transcription::transformation::{
use crate::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
@ -20,15 +20,15 @@ pub struct AudioTranscriptionRequest<'a> {
#[derive(Clone)]
pub struct ProviderAudioTranscriptionRequest {
pub(super) model: String,
pub(super) custom_llm_provider: String,
pub(super) config: &'static dyn BaseAudioTranscriptionConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) auth: AudioTranscriptionAuth,
pub(super) optional_params: Map<String, Value>,
pub(super) timeout: Option<Duration>,
pub model: String,
pub custom_llm_provider: String,
pub config: &'static dyn BaseAudioTranscriptionConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub auth: AudioTranscriptionAuth,
pub optional_params: Map<String, Value>,
pub timeout: Option<Duration>,
}
impl ProviderAudioTranscriptionRequest {

View file

@ -1,9 +1,9 @@
use serde_json::{Map, Value};
use crate::llms::anthropic::experimental_pass_through::messages::transformation::{
use crate::anthropic::experimental_pass_through::messages::transformation::{
ANTHROPIC_MESSAGES_CONFIG, AnthropicMessagesConfig, non_empty,
};
use crate::llms::base_llm::anthropic_messages::transformation::{
use crate::base_llm::anthropic_messages::transformation::{
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
};
use crate::messages::Error;

View file

@ -0,0 +1 @@
pub mod anthropic;

View file

@ -0,0 +1 @@
pub mod transformation;

View file

@ -0,0 +1 @@
pub mod transformation;

View file

@ -1,7 +1,7 @@
use serde_json::{Map, Value};
use crate::chat_completions::Error;
use crate::chat_completions::types::{
use crate::chat::Error;
use crate::chat::types::{
ChatCompletionsResponse, ChatMessage, ChatMessageContent, ProviderChatRequestData,
ProviderChatResponseData,
};

View file

@ -0,0 +1,3 @@
pub mod anthropic_messages;
pub mod audio_transcription;
pub mod chat;

View file

@ -1,11 +1,11 @@
use serde_json::{Map, Value, json};
use crate::audio_transcription::Error;
use crate::audio_transcription::json_type_name;
use crate::audio_transcription::types::{
AudioTranscriptionRequestData, AudioTranscriptionResponseData,
};
use crate::http_utils::json_type_name;
use crate::llms::base_llm::audio_transcription::transformation::{
use crate::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
use litellm_auth_aws::constants::{BEDROCK_RUNTIME_ENDPOINT_TEMPLATE, BEDROCK_SERVICE};

View file

@ -1,16 +1,16 @@
use serde_json::{Map, Value, json};
use crate::chat_completions::Error;
use crate::chat_completions::conversation::{Conversation, TurnRole, build_conversation};
use crate::chat_completions::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat_completions::types::{
use crate::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use crate::chat::Error;
use crate::chat::conversation::{Conversation, TurnRole, build_conversation};
use crate::chat::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse,
ChatCompletionsUsage, ChatMessage, ChatMessageContent, ProviderChatRequestData,
ProviderChatResponseData,
};
use crate::llms::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use litellm_auth_aws::constants::{AWS_BEARER_TOKEN_BEDROCK, BEDROCK_RUNTIME_ENDPOINT_TEMPLATE};
use litellm_auth_aws::{bedrock_model_id_and_region, resolve_bedrock_region};

View file

@ -1,7 +1,7 @@
use serde_json::json;
use super::*;
use crate::chat_completions::Error;
use crate::chat::Error;
fn messages(value: Value) -> Vec<ChatMessage> {
serde_json::from_value(value).expect("valid messages")

View file

@ -11,7 +11,7 @@
//! accepts; anything richer is declined upstream by the capability gate.
use super::types::{ChatMessage, ChatMessageContent};
use crate::constants::EMPTY_TEXT_PLACEHOLDER;
use crate::chat::EMPTY_TEXT_PLACEHOLDER;
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum TurnRole {

View file

@ -0,0 +1,21 @@
use thiserror::Error;
pub const EMPTY_TEXT_PLACEHOLDER: &str = " ";
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error("unsupported: {0}")]
Unsupported(&'static str),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub mod conversation;
pub mod response_utils;
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
use crate::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
/// A `/chat/completions` call as it crosses into the core.
///
@ -22,26 +22,26 @@ pub struct ChatCompletionsRequest<'a> {
pub timeout: Option<Duration>,
}
pub(super) struct ResolvedChatCompletionsRequest<'a> {
pub(super) model: String,
pub(super) config: &'static dyn BaseConfig,
pub(super) messages: Vec<ChatMessage>,
pub(super) optional_params: Map<String, Value>,
pub(super) api_key: Option<&'a str>,
pub(super) api_base: Option<&'a str>,
pub(super) extra_headers: Option<Map<String, Value>>,
pub(super) timeout: Option<Duration>,
pub struct ResolvedChatCompletionsRequest<'a> {
pub model: String,
pub config: &'static dyn BaseConfig,
pub messages: Vec<ChatMessage>,
pub optional_params: Map<String, Value>,
pub api_key: Option<&'a str>,
pub api_base: Option<&'a str>,
pub extra_headers: Option<Map<String, Value>>,
pub timeout: Option<Duration>,
}
pub(super) struct ProviderChatCompletionsRequest {
pub(super) model: String,
pub(super) config: &'static dyn BaseConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) auth: ChatCompletionsAuth,
pub(super) optional_params: Map<String, Value>,
pub(super) timeout: Option<Duration>,
pub struct ProviderChatCompletionsRequest {
pub model: String,
pub config: &'static dyn BaseConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub auth: ChatCompletionsAuth,
pub optional_params: Map<String, Value>,
pub timeout: Option<Duration>,
}
/// The provider-shaped request body a config produces. Named rather than a bare

View file

@ -0,0 +1,8 @@
pub mod anthropic;
pub mod audio_transcription;
pub mod azure_ai;
pub mod base_llm;
pub mod bedrock;
pub mod chat;
pub mod messages;
pub mod provider_resolution;

View file

@ -0,0 +1,17 @@
use thiserror::Error;
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error("unsupported: {0}")]
Unsupported(&'static str),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
pub struct MessagesRequest<'a> {
pub model: &'a str,
@ -15,14 +15,14 @@ pub struct MessagesRequest<'a> {
pub timeout: Option<Duration>,
}
pub(super) struct ProviderMessagesRequest {
pub(super) provider: String,
pub(super) model: String,
pub(super) config: &'static dyn BaseAnthropicMessagesConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) timeout: Option<Duration>,
pub struct ProviderMessagesRequest {
pub provider: String,
pub model: String,
pub config: &'static dyn BaseAnthropicMessagesConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub timeout: Option<Duration>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]

View file

@ -0,0 +1,33 @@
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct CustomLlmProvider<'a> {
pub model: &'a str,
pub custom_llm_provider: &'a str,
}
pub fn get_custom_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
) -> Option<CustomLlmProvider<'a>> {
if let Some(custom_llm_provider) = custom_llm_provider.filter(|provider| !provider.is_empty()) {
return Some(CustomLlmProvider {
model: strip_custom_llm_provider_prefix(model, custom_llm_provider),
custom_llm_provider,
});
}
let (custom_llm_provider, model) = model.split_once('/')?;
if custom_llm_provider.is_empty() || model.is_empty() {
return None;
}
Some(CustomLlmProvider {
model,
custom_llm_provider,
})
}
fn strip_custom_llm_provider_prefix<'a>(model: &'a str, custom_llm_provider: &str) -> &'a str {
model
.strip_prefix(custom_llm_provider)
.and_then(|model| model.strip_prefix('/'))
.unwrap_or(model)
}

View file

@ -69184,6 +69184,27 @@
"supports_reasoning": true,
"supports_vision": true
},
"typesafe/jev-1.13.0": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-latest": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-preview": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"wandb/zai-org/GLM-5.3-Flash": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_token": 1.5e-07,

View file

@ -208,6 +208,7 @@ LAZY_FEATURES: Final[tuple[LazyFeature, ...]] = (
"/nvidia_nim/",
"/openai/",
"/openai_passthrough/",
"/typesafe/",
"/vertex-ai/",
"/vertex_ai/",
"/vllm/",

View file

@ -20373,6 +20373,96 @@
]
}
},
"/typesafe/{endpoint}": {
"get": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__get",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
},
"post": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__post",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
}
},
"/vertex_ai/discovery/{endpoint}": {
"delete": {
"description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`",

View file

@ -483,6 +483,7 @@ class LiteLLMRoutes(enum.Enum):
"/eu.assemblyai",
"/vllm",
"/mistral",
"/typesafe",
"/milvus",
"/gigachat",
"/watsonx",

View file

@ -17,6 +17,7 @@ if TYPE_CHECKING:
AUTO_ROUTER_LICENSE_FEATURE: Final = "auto_router"
LICENSE_ALL_FEATURES: Final = "*"
AUTO_ROUTER_LICENSE_REMEDY: Final = "A LiteLLM license with the 'auto_router' feature lifts the limit."
@ -153,17 +154,21 @@ class LicenseCheck:
return False
return team_count > _max_teams_in_license
def grants_feature(self, feature: str) -> bool:
if self.airgapped_license_data is None:
return False
allowed_features: Final = self.airgapped_license_data.get("allowed_features")
granted: Final = allowed_features if isinstance(allowed_features, list) else (allowed_features,)
return feature in granted or LICENSE_ALL_FEATURES in granted
def auto_router_capability_limit(self) -> int | None:
"""
How many auto-routers may claim each gated classifier or customization capability:
unlimited (None) only when the signed license lists the auto_router
feature, otherwise one per capability. A license verified through the API carries no
feature list, so it does not lift the limit either.
unlimited (None) only when the signed license lists the auto_router feature or the
"*" wildcard that grants every feature, otherwise one per capability. A license verified
through the API carries no feature list, so it does not lift the limit either.
"""
if self.airgapped_license_data is None:
return 1
allowed_features: Final = self.airgapped_license_data.get("allowed_features")
if isinstance(allowed_features, list) and AUTO_ROUTER_LICENSE_FEATURE in allowed_features:
if self.grants_feature(AUTO_ROUTER_LICENSE_FEATURE):
return None
return 1

View file

@ -569,15 +569,15 @@ lite --base-url https://your-proxy.example.com configure claude --api-key sk-...
claude
```
The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-<UTF-8 hex of the group name>` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute up` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control
The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-<UTF-8 hex of the group name>` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute start` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control
Plain `lite configure`, with no agent named, asks which agents to wire and which gateway model each starts on, picked from `/v1/models` with a type-to-filter prompt. All choices and selected config files are checked before the first settings write. If a later filesystem write fails, the output identifies each agent already configured and its undo command
What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute up` session holds a backup, and that check comes before any request
What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute start` session holds a backup, and that check comes before any request
#### Routed model and savings in the status line
`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute up` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline:
`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute start` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline:
```
Routed to: claude-haiku-4-5 -63% vs Claude Opus 5
@ -597,7 +597,7 @@ After upgrading the CLI, rerun your original `lite configure claude` command wit
#### Install the CLI
`lite autoroute up` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
`lite autoroute start` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
```bash
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install.sh | sh
@ -610,7 +610,7 @@ curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch-or-commit>/
LITELLM_CLI_REF=<branch-or-commit> sh
```
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute up`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute start`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
Point the CLI at your real proxy and key before running any `lite model-groups` or `lite autoroute` command -- like every other command in this CLI, they read `LITELLM_PROXY_URL`/`LITELLM_PROXY_API_KEY` (or `--base-url`/`--api-key`), no `lite login` required:
@ -637,44 +637,46 @@ An interactive wizard. It runs the same model-group discovery as above, splits t
The wizard writes the result to `~/.litellm/autorouter/config.yaml` with `0600` permissions, since the file embeds your real proxy API key. Every model referenced anywhere in that config -- tier targets, the classifier model, the embedding model -- becomes its own `litellm_proxy/<model-name>` deployment whose `api_base` and `api_key` point back at your real proxy. That is the trick that keeps your real proxy's config untouched: every actual network call this generates, whether it is the routed completion, an LLM-classifier call, or an embedding call, forwards transparently through your real, already-running proxy with your real key.
You do not need to tell Claude Code to request `autorouter` by name yourself: `lite autoroute up` also sets the top-level `model` and `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`, `ANTHROPIC_DEFAULT_OPUS_MODEL` and `ANTHROPIC_DEFAULT_FABLE_MODEL` to `autorouter` in `~/.claude/settings.json` (and `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when missing, like every other wiring), so every one of Claude Code's own model tiers requests it directly regardless of `/model` or whatever it defaults to otherwise. (A bare `model_name: "*"` deployment looks like the obvious way to catch any request instead, but litellm's Router looks up auto-router deployments by the literal requested model string with no wildcard resolution, so a `"*"` entry would never actually match real traffic -- these env var overrides are what makes it work.)
You do not need to tell Claude Code to request `autorouter` by name yourself: `lite autoroute start` also sets the top-level `model` and `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`, `ANTHROPIC_DEFAULT_OPUS_MODEL` and `ANTHROPIC_DEFAULT_FABLE_MODEL` to `autorouter` in `~/.claude/settings.json` (and `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when missing, like every other wiring), so every one of Claude Code's own model tiers requests it directly regardless of `/model` or whatever it defaults to otherwise. (A bare `model_name: "*"` deployment looks like the obvious way to catch any request instead, but litellm's Router looks up auto-router deployments by the literal requested model string with no wildcard resolution, so a `"*"` entry would never actually match real traffic -- these env var overrides are what makes it work.)
You must run `configure` at least once before `up`; running `up` first fails with a clear error telling you to configure first.
You must run `configure` at least once before `start`; running `start` first fails with a clear error telling you to configure first.
#### Launch the Ephemeral Auto-Router Proxy
```bash
lite autoroute up
lite autoroute start
```
Starts a local, throwaway litellm proxy on `127.0.0.1:5483` (override with `--port`), running the config `configure` generated, with a self-issued API key baked in (your real proxy key never leaves the generated config -- it only appears there, forwarding to your real proxy). Both the port and the key are stable across runs: the key is minted once, persisted inside the generated config, and reused by every later `up` (and carried forward when you re-run `configure`), so anything you configured against one session keeps working in the next. If the port is already taken, `up` refuses with a clear error instead of silently moving to another one. It waits for the ephemeral proxy to report healthy, then patches `~/.claude/settings.json` the same way `lite up` does, except with a static `ANTHROPIC_AUTH_TOKEN` env var instead of an `apiKeyHelper`, since this key is self-issued rather than something needing SSO refresh. Any `claude` session started afterward, from any terminal, routes through the ephemeral proxy.
Starts a local, throwaway litellm proxy on `127.0.0.1:5483` (override with `--port`), running the config `configure` generated, with a self-issued API key baked in (your real proxy key never leaves the generated config -- it only appears there, forwarding to your real proxy). Both the port and the key are stable across runs: the key is minted once, persisted inside the generated config, and reused by every later `start` (and carried forward when you re-run `configure`), so anything you configured against one session keeps working in the next. If the port is already taken, `start` refuses with a clear error instead of silently moving to another one. It waits for the ephemeral proxy to report healthy, then patches `~/.claude/settings.json` the same way `lite up` does, except with a static `ANTHROPIC_AUTH_TOKEN` env var instead of an `apiKeyHelper`, since this key is self-issued rather than something needing SSO refresh. Any `claude` session started afterward, from any terminal, routes through the ephemeral proxy.
`lite autoroute up` runs in the foreground and streams the ephemeral proxy's own log file into your terminal, so you can watch its routing decisions -- which tier and model got picked for each request -- as you use Claude Code normally. Press Ctrl-C (or send SIGTERM) to stop it; this kills the child proxy process and restores your original Claude Code settings, in that order.
`lite autoroute start` runs in the foreground and streams the ephemeral proxy's own log file into your terminal, so you can watch its routing decisions -- which tier and model got picked for each request -- as you use Claude Code normally. Press Ctrl-C (or send SIGTERM) to stop it; this kills the child proxy process and restores your original Claude Code settings, in that order.
#### Recover From an Unclean Shutdown
```bash
lite autoroute down
lite autoroute stop
```
If the `lite autoroute up` process dies uncleanly -- `kill -9`, a crash -- rather than being stopped with Ctrl-C, `down` is the manual recovery path: it kills any leftover ephemeral proxy process found via a recorded pid file and restores Claude Code's settings from whatever backup is on disk.
If the `lite autoroute start` process dies uncleanly -- `kill -9`, a crash -- rather than being stopped with Ctrl-C, `stop` is the manual recovery path: it kills any leftover ephemeral proxy process found via a recorded pid file and restores Claude Code's settings from whatever backup is on disk.
#### Example
```bash
lite autoroute configure
lite autoroute up
lite autoroute start
# use Claude Code as normal in another terminal; routing decisions stream live
lite autoroute down # only needed if `up` was killed uncleanly instead of Ctrl-C'd
lite autoroute stop # only needed if `start` was killed uncleanly instead of Ctrl-C'd
```
The previous names, `lite autoroute up` and `lite autoroute down`, still work as hidden aliases of `start` and `stop`: each prints a deprecation notice on stderr and will be removed in a future release
#### Caveats
Adaptive mode's learned state does not persist across `lite autoroute up` sessions -- there is no local database, so every session starts adaptive selection cold. A Claude Code session already running before `up` started, or still running when it stops, keeps whatever settings it loaded at its own startup; like `lite up`, this is a one-time file patch and restore, not a live traffic interceptor. Only Claude Code is supported, for the same reason as `lite up`: no other supported agent (for example Cursor) has an equivalent hot-patchable config file.
Adaptive mode's learned state does not persist across `lite autoroute start` sessions -- there is no local database, so every session starts adaptive selection cold. A Claude Code session already running before `start` ran, or still running when it stops, keeps whatever settings it loaded at its own startup; like `lite up`, this is a one-time file patch and restore, not a live traffic interceptor. Only Claude Code is supported, for the same reason as `lite up`: no other supported agent (for example Cursor) has an equivalent hot-patchable config file.
A session that outlives `up` (or is still running the moment you stop it) keeps sending requests, master key included, to that now-freed loopback port until you restart it. Once the ephemeral proxy process exits, nothing stops another local account on the same machine from binding that same port and receiving those requests instead -- and since the port is a fixed, predictable default and the master key is a static value that persists across sessions (unlike `lite up`'s `apiKeyHelper`, which is re-resolved per request), whoever receives them gets a live-looking token along with the prompt content. Restart any Claude Code session before you consider the machine clean, run `lite autoroute down` promptly rather than leaving a stopped session's settings patched, and do not run `lite autoroute up` on a shared or multi-tenant host. To rotate the persisted key, delete the `master_key` line from `~/.litellm/autorouter/config.yaml`; the next `up` mints a fresh one (deleting the whole file works too, but then `configure` must be re-run first).
A session that outlives `start` (or is still running the moment you stop it) keeps sending requests, master key included, to that now-freed loopback port until you restart it. Once the ephemeral proxy process exits, nothing stops another local account on the same machine from binding that same port and receiving those requests instead -- and since the port is a fixed, predictable default and the master key is a static value that persists across sessions (unlike `lite up`'s `apiKeyHelper`, which is re-resolved per request), whoever receives them gets a live-looking token along with the prompt content. Restart any Claude Code session before you consider the machine clean, run `lite autoroute stop` promptly rather than leaving a stopped session's settings patched, and do not run `lite autoroute start` on a shared or multi-tenant host. To rotate the persisted key, delete the `master_key` line from `~/.litellm/autorouter/config.yaml`; the next `start` mints a fresh one (deleting the whole file works too, but then `configure` must be re-run first).
Do not run `lite up` and `lite autoroute up` at the same time. Each patches `~/.claude/settings.json` and keeps its own separate backup, with no coordination between them: whichever one you stop or crash out of last is the one whose backup gets restored, which can silently leave the *other* mode's settings (a static master key and a now-dead loopback URL, or a stale `apiKeyHelper`) active. Run `lite down` or `lite autoroute down` (whichever applies) before switching to the other mode.
Do not run `lite up` and `lite autoroute start` at the same time. Each patches `~/.claude/settings.json` and keeps its own separate backup, with no coordination between them: whichever one you stop or crash out of last is the one whose backup gets restored, which can silently leave the *other* mode's settings (a static master key and a now-dead loopback URL, or a stale `apiKeyHelper`) active. Run `lite down` or `lite autoroute stop` (whichever applies) before switching to the other mode.
## Environment Variables

View file

@ -51,7 +51,7 @@ def _ensure_master_key() -> str:
The generated config is the single home of the key: the proxy server authenticates against
general_settings.master_key only (a key under litellm_settings is silently ignored, which
would leave the ephemeral proxy with no real auth), and the file is written 0600 via
secure_create. Reusing that persisted value keeps the key stable across `up` runs, so a
secure_create. Reusing that persisted value keeps the key stable across `start` runs, so a
client configured against one session keeps working in the next.
"""
with open(CONFIG_PATH, "r") as f:
@ -88,15 +88,18 @@ def configure(ctx: click.Context) -> None:
run_configure_wizard(ctx)
@autoroute_group.command("up")
@click.option(
_PORT_OPTION: Final = click.option(
"--port",
type=click.IntRange(1, 65535),
default=DEFAULT_AUTOROUTE_PORT,
show_default=True,
help="Loopback port for the ephemeral proxy; stable across runs so configured clients keep working.",
)
def up(port: int) -> None:
@autoroute_group.command("start")
@_PORT_OPTION
def start(port: int) -> None:
"""Launch the ephemeral auto-router proxy and route Claude Code through it"""
if not CONFIG_PATH.exists():
raise click.ClickException("No config found. Run `lite autoroute configure` first.")
@ -104,7 +107,7 @@ def up(port: int) -> None:
missing: Final = missing_proxy_runtime_modules()
if missing:
raise click.ClickException(
"lite autoroute up launches a local litellm proxy, which needs the proxy runtime that the "
"lite autoroute start launches a local litellm proxy, which needs the proxy runtime that the "
f"thin `litellm[cli]` install does not include (missing: {', '.join(missing)}). Install the "
"proxy runtime with `uv tool install --force 'litellm[proxy]'`, or to QA a branch, "
"`curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch>/scripts/install.sh | "
@ -117,14 +120,14 @@ def up(port: int) -> None:
raise click.ClickException(str(e))
if existing_pid is not None and is_running(existing_pid.pid):
raise click.ClickException(
"An ephemeral proxy is already running (lite autoroute up looks already active). "
"Run `lite autoroute down` first."
"An ephemeral proxy is already running (lite autoroute start looks already active). "
"Run `lite autoroute stop` first."
)
if AUTOROUTE_BACKUP_PATH.exists():
raise click.ClickException(
f"{AUTOROUTE_BACKUP_PATH} already exists -- `lite autoroute up` looks like it's already "
"running (or crashed without cleanup). Run `lite autoroute down` first."
f"{AUTOROUTE_BACKUP_PATH} already exists -- `lite autoroute start` looks like it's already "
"running (or crashed without cleanup). Run `lite autoroute stop` first."
)
if port == 4000:
@ -135,8 +138,8 @@ def up(port: int) -> None:
if not is_port_available(port):
raise click.ClickException(
f"Port {port} on 127.0.0.1 is already in use. If a previous `lite autoroute up` is still "
"running or crashed, run `lite autoroute down`; otherwise pick a different port with --port."
f"Port {port} on 127.0.0.1 is already in use. If a previous `lite autoroute start` is still "
"running or crashed, run `lite autoroute stop`; otherwise pick a different port with --port."
)
master_key: Final = _ensure_master_key()
@ -196,7 +199,7 @@ def up(port: int) -> None:
click.echo("\nStopped ephemeral proxy and restored Claude Code settings.")
click.echo(
f"Restart any Claude Code session still open from this session, or another local account could "
f"bind the now-free port {port} and receive its requests. Do not use `lite autoroute up` on a "
f"bind the now-free port {port} and receive its requests. Do not use `lite autoroute start` on a "
f"shared or multi-tenant host."
)
@ -214,13 +217,13 @@ def up(port: int) -> None:
_teardown()
@autoroute_group.command("down")
def down() -> None:
@autoroute_group.command("stop")
def stop() -> None:
"""Restore Claude Code settings and stop a leftover ephemeral proxy, if any"""
try:
record: PidRecord | None = read_pid_record()
except ClaudeSettingsError as e:
# down is the crash-recovery path -- a corrupt pid record must not block it; clear the
# stop is the crash-recovery path -- a corrupt pid record must not block it; clear the
# unusable record and keep going rather than leaving the user with no way to clean up.
click.echo(f"{e} Clearing it and continuing cleanup.", err=True)
record = None
@ -238,7 +241,34 @@ def down() -> None:
elif restored.existed:
click.echo(f"Restored {CLAUDE_SETTINGS_PATH} to its original contents.")
else:
click.echo(f"Removed {CLAUDE_SETTINGS_PATH} (it did not exist before `lite autoroute up`).")
click.echo(f"Removed {CLAUDE_SETTINGS_PATH} (it did not exist before `lite autoroute start`).")
AUTOROUTE_ALIAS_DEPRECATION_NOTICE: Final = (
"`lite autoroute {retired}` is deprecated and will be removed in a future release; "
"run `lite autoroute {current}` instead, it takes the same options."
)
def _warn_deprecated_alias(retired: str, current: str) -> None:
click.secho(AUTOROUTE_ALIAS_DEPRECATION_NOTICE.format(retired=retired, current=current), err=True, fg="yellow")
@autoroute_group.command("up", hidden=True)
@_PORT_OPTION
@click.pass_context
def up(ctx: click.Context, port: int) -> None:
"""Deprecated alias of `lite autoroute start`"""
_warn_deprecated_alias("up", "start")
ctx.invoke(start, port=port)
@autoroute_group.command("down", hidden=True)
@click.pass_context
def down(ctx: click.Context) -> None:
"""Deprecated alias of `lite autoroute stop`"""
_warn_deprecated_alias("down", "stop")
ctx.invoke(stop)
__all__ = ["autoroute_group"]

View file

@ -214,7 +214,7 @@ def build_generated_proxy_config(config: AutorouteConfig, master_key: str) -> di
def master_key_from_config(config: dict[str, JsonValue]) -> str | None:
"""The master key persisted in a generated config, or None when absent or blank.
Single definition of "this config already has a usable key", shared by `up` (reuse
Single definition of "this config already has a usable key", shared by `start` (reuse
instead of minting) and the configure wizard (carry the key forward on rewrite) so the
two sites can never disagree on what counts as one. Returned verbatim, never stripped:
the proxy authenticates against the exact bytes under general_settings.master_key, so a

View file

@ -43,12 +43,12 @@ _PROXY_RUNTIME_MODULES: tuple[str, ...] = ("fastapi", "uvicorn", "backoff", "orj
def missing_proxy_runtime_modules() -> tuple[str, ...]:
"""Proxy-server modules that ``lite autoroute up`` needs but the thin CLI install lacks.
"""Proxy-server modules that ``lite autoroute start`` needs but the thin CLI install lacks.
``launch_proxy`` runs the full ``litellm.proxy.proxy_cli`` server, whose dependencies live in
the ``proxy`` extra, not the ``cli`` extra that installs the ``lite`` command. On a thin
``litellm[cli]`` install the subprocess dies with a bare ``ModuleNotFoundError``; detecting the
gap here lets ``up`` fail with an actionable message instead.
gap here lets ``start`` fail with an actionable message instead.
"""
return tuple(name for name in _PROXY_RUNTIME_MODULES if importlib.util.find_spec(name) is None)

View file

@ -94,7 +94,7 @@ def _load_persisted_master_key(config_path: Path) -> str | None:
"""The master key from an existing generated config, so a rewrite carries it forward.
Lenient on a missing or corrupt file: configure is the regeneration path, so it must
succeed from any prior state; a key that cannot be read is simply not carried and `up`
succeed from any prior state; a key that cannot be read is simply not carried and `start`
mints a fresh one.
"""
if not config_path.exists():

View file

@ -1,6 +1,6 @@
"""Shared handling of Claude Code's ~/.claude/settings.json.
`lite up` and `lite autoroute up` patch this file temporarily and restore it on
`lite up` and `lite autoroute start` patch this file temporarily and restore it on
exit; `lite configure claude` patches it persistently and records how to undo it.
All of them need the same merge, and `up` already imports from `auth`, so the
shared parts live here rather than in any one command module. The credential is
@ -88,7 +88,7 @@ class SettingsFileOwner:
SETTINGS_FILE_OWNERS: Final = (
SettingsFileOwner(BACKUP_PATH, "lite up", "lite down"),
SettingsFileOwner(AUTOROUTE_BACKUP_PATH, "lite autoroute up", "lite autoroute down"),
SettingsFileOwner(AUTOROUTE_BACKUP_PATH, "lite autoroute start", "lite autoroute stop"),
)
_SETTINGS_ADAPTER: Final = TypeAdapter(dict[str, JsonValue])
@ -111,7 +111,7 @@ def _is_default_settings_file(settings_path: Path) -> bool:
def settings_file_owners(settings_path: Path) -> tuple[SettingsFileOwner, ...]:
"""The commands whose backups guard settings_path: `lite up` and `lite autoroute up` only ever manage the default file."""
"""The commands whose backups guard settings_path: `lite up` and `lite autoroute start` only ever manage the default file."""
return SETTINGS_FILE_OWNERS if _is_default_settings_file(settings_path) else ()
@ -240,7 +240,7 @@ def _env_object(settings: Mapping[str, JsonValue], path: Path) -> Mapping[str, J
def refuse_while_owned(settings_path: Path, owners: Sequence[SettingsFileOwner]) -> None:
"""Refuse while `lite up` or `lite autoroute up` holds a backup it will restore over any write; a
"""Refuse while `lite up` or `lite autoroute start` holds a backup it will restore over any write; a
purely local check, so commands run it before any login prompt or request."""
for owner in owners:
if owner.backup_path.exists():
@ -262,7 +262,7 @@ def _write_target(settings_path: Path) -> Path:
def write_claude_settings(settings_path: Path, settings: Mapping[str, JsonValue]) -> None:
"""The one way a settings document lands on disk: staged owner-only beside the target and renamed into
place, through a symlink rather than over it. Every writer (`configure`, `up`, `autoroute up` and the
place, through a symlink rather than over it. Every writer (`configure`, `up`, `autoroute start` and the
restores) may be carrying the credential, so none creates the file under the umask or truncates it."""
target: Final = _write_target(settings_path)
try:
@ -341,7 +341,7 @@ def merge_claude_settings(
an apiKeyHelper) are removed, since Claude Code given two credentials may send the wrong one.
ENABLE_TOOL_SEARCH and CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY get their defaults only when
missing. `default_model` is the top-level `model` and env.ANTHROPIC_MODEL (see StartOn);
`tier_model` is `lite autoroute up`'s knob that points every ANTHROPIC_DEFAULT_*_MODEL at one
`tier_model` is `lite autoroute start`'s knob that points every ANTHROPIC_DEFAULT_*_MODEL at one
group. Apart from those tier keys, exactly OWNED_PATHS are touched.
"""
raw_env: Final = settings.get(ENV_KEY, {})

View file

@ -56,7 +56,7 @@ _CLAUDE_CODE_VIEW: Final = MappingProxyType(
_MODEL_OPTION_HELP: Final = (
f"Proxy model to set as {STARTING_MODEL_ROLE}. Must be listed on /v1/models for the key; without it, "
"Claude Code keeps its own default and a pin an earlier configure made is let go of. Nothing pins Claude "
"Code's sub-agent or background tiers; `lite autoroute up` is the mode that does."
"Code's sub-agent or background tiers; `lite autoroute start` is the mode that does."
)

View file

@ -22,7 +22,7 @@ from litellm.types.llms.openai import (
BaseLiteLLMOpenAIResponseObject,
ResponsesAPIResponse,
)
from litellm.types.utils import CallTypesLiteral, LLMResponseTypes, SpecialEnums
from litellm.types.utils import ADDRESSED_RESPONSE_ID_FIELD, CallTypesLiteral, LLMResponseTypes, SpecialEnums
if TYPE_CHECKING:
from litellm.caching.caching import DualCache
@ -32,7 +32,6 @@ if TYPE_CHECKING:
_RESPONSES_API_PROVIDER_PREFIX: Final = "/openai"
_RESPONSES_API_CREATE_ROUTES: Final = frozenset({"/v1/responses", "/responses"})
_ADDRESSED_RESPONSE_ID_KEY: Final = "_litellm_addressed_response_id"
_UNMANAGED_RESPONSE_ID_DETAIL: Final = (
"Forbidden. This response id was not issued by this proxy, so the proxy cannot tell who owns it. "
"To let keys address responses this proxy did not issue, set "
@ -132,7 +131,7 @@ class ResponsesIDSecurity(CustomLogger):
if call_type not in responses_api_call_types:
return None
addressed_id_field: Final = "previous_response_id" if call_type == "aresponses" else "response_id"
retained_id: Final = data.get(_ADDRESSED_RESPONSE_ID_KEY)
retained_id: Final = data.get(ADDRESSED_RESPONSE_ID_FIELD)
addressed_id: Final = (
retained_id if isinstance(retained_id, str) and retained_id else data.get(addressed_id_field)
)
@ -140,7 +139,7 @@ class ResponsesIDSecurity(CustomLogger):
return data
authorized_id: Final = self._authorize_response_id(addressed_id, user_api_key_dict)
data[addressed_id_field] = authorized_id
data[_ADDRESSED_RESPONSE_ID_KEY] = addressed_id
data[ADDRESSED_RESPONSE_ID_FIELD] = addressed_id
return data
def _authorize_response_id(

View file

@ -586,35 +586,38 @@ async def _upsert_budget_and_membership(
)
return
create_data: Final[dict[str, Any]] = {
"created_by": user_api_key_dict.user_id or "",
"updated_by": user_api_key_dict.user_id or "",
}
seed_row_id: Final = (
seeds_temp_budget: Final = "temp_budget_increase" in write_data or "temp_budget_expiry" in write_data
source_row_id: Final = (
existing_budget_id
if is_shared_default
else team_default_budget_id
if team_default_budget_id is not None
and ("temp_budget_increase" in write_data or "temp_budget_expiry" in write_data)
if team_default_budget_id is not None and seeds_temp_budget
else None
)
if seed_row_id is not None:
seed_row: Final = await tx.litellm_budgettable.find_unique(where={"budget_id": seed_row_id})
if seed_row is not None:
seed_dict: Final = seed_row.model_dump()
for field in _TEAM_MEMBER_BUDGET_LIMIT_FIELDS:
value = seed_dict.get(field)
if field == "max_budget" and value == 0 and not is_shared_default:
continue
if _is_set_budget_value(value):
create_data[field] = value
source_row: Final = (
await tx.litellm_budgettable.find_unique(where={"budget_id": source_row_id})
if source_row_id is not None
else None
)
source: Final[Mapping[str, Any]] = source_row.model_dump() if source_row is not None else MappingProxyType({})
create_data.update(write_data)
def _seeds(field: str) -> bool:
if field == "max_budget" and source.get(field) == 0 and not is_shared_default:
return False
return _is_set_budget_value(source.get(field))
if create_data.get("budget_duration") is not None:
create_data["budget_reset_at"] = get_budget_reset_time(budget_duration=create_data["budget_duration"])
else:
create_data: Final[dict[str, Any]] = { # mutable-ok: Prisma create payloads are dict-shaped
"created_by": user_api_key_dict.user_id or "",
"updated_by": user_api_key_dict.user_id or "",
**MappingProxyType({f: source[f] for f in _TEAM_MEMBER_BUDGET_LIMIT_FIELDS if _seeds(f)}),
**write_data,
}
# Restarting an inherited window on an unrelated edit hands the member a free period.
carried: Final = source.get("budget_reset_at") if "budget_duration" not in budget_patch else None
if carried is not None:
create_data["budget_reset_at"] = carried
if create_data.get("budget_reset_at") is None:
create_data.pop("budget_reset_at", None)
if not _has_meaningful_budget_limit(create_data):

View file

@ -2,7 +2,7 @@
from typing import Annotated, Final
from fastapi import APIRouter, Depends
from fastapi import APIRouter, Depends, Header
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth
@ -108,6 +108,12 @@ async def bulk_update_team_member_budgets_action(
team_id: str,
data: BulkTeamMemberBudgetUpdateRequest,
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
litellm_changed_by: Annotated[
str | None,
Header(
description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
),
] = None,
) -> BulkTeamMemberBudgetUpdateResponse:
"""
Set per-member limits for up to 500 members of one team in one call. Same
@ -135,7 +141,7 @@ async def bulk_update_team_member_budgets_action(
```
"""
try:
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
from litellm.proxy.proxy_server import litellm_proxy_admin_name, prisma_client, user_api_key_cache
if prisma_client is None:
raise ManagementProblem(
@ -153,6 +159,8 @@ async def bulk_update_team_member_budgets_action(
user_api_key_dict=user_api_key_dict,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
litellm_proxy_admin_name=litellm_proxy_admin_name,
litellm_changed_by=litellm_changed_by,
)
return BulkTeamMemberBudgetUpdateResponse(data=results)

View file

@ -7,11 +7,20 @@ cap never moves another member's.
"""
from collections.abc import Sequence
from datetime import timedelta
from datetime import datetime, timedelta
from types import MappingProxyType
from typing import TYPE_CHECKING, Final
from litellm.proxy._types import LiteLLM_TeamTable, LitellmUserRoles, Member, UserAPIKeyAuth
from pydantic import BaseModel, ConfigDict
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.proxy._types import (
LiteLLM_TeamTable,
LitellmTableNames,
LitellmUserRoles,
Member,
UserAPIKeyAuth,
)
from litellm.proxy.auth.auth_checks import invalidate_team_member_spend_state
from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
from litellm.proxy.db.routing_prisma_wrapper import WriterPinnedClient
@ -21,6 +30,7 @@ from litellm.proxy.management_endpoints.common_utils import (
_upsert_budget_and_membership, # pyright: ignore[reportPrivateUsage] # the single-member write, shared so the two surfaces cannot drift
member_budget_patch,
)
from litellm.proxy.management_helpers.audit_logs import create_object_audit_log
from litellm.proxy.management_helpers.bulk_user_deletion import (
_duplicate_member_indexes, # pyright: ignore[reportPrivateUsage] # same duplicate rule as members/bulk_delete
_eq_filter, # pyright: ignore[reportPrivateUsage] # same prisma filter shape as members/bulk_delete
@ -77,6 +87,54 @@ async def _shared_budget_ids(tx: "Prisma", budget_ids: frozenset[str]) -> frozen
return frozenset(budget_id for budget_id in budget_ids if sum(1 for row in rows if row.budget_id == budget_id) > 1)
class _AuditedMemberBudget(BaseModel):
"""One member's limits as the audit log's before/after values record them."""
model_config = ConfigDict(frozen=True)
user_id: str
budget_id: str | None = None
max_budget: float | None = None
tpm_limit: int | None = None
rpm_limit: int | None = None
budget_duration: str | None = None
budget_reset_at: datetime | None = None
allowed_models: tuple[str, ...] | None = None
class _AuditedMemberBudgets(BaseModel):
"""The audit-log columns hold a JSON object, so the per-member list is nested under a key."""
model_config = ConfigDict(frozen=True)
team_member_budgets: tuple[_AuditedMemberBudget, ...]
def _audited_member_budget(row: "prisma_models.LiteLLM_TeamMembership") -> _AuditedMemberBudget:
budget: Final = row.litellm_budget_table
if budget is None:
return _AuditedMemberBudget(user_id=row.user_id, budget_id=row.budget_id)
return _AuditedMemberBudget(
user_id=row.user_id,
budget_id=row.budget_id,
max_budget=budget.max_budget,
tpm_limit=budget.tpm_limit,
rpm_limit=budget.rpm_limit,
budget_duration=budget.budget_duration,
budget_reset_at=budget.budget_reset_at,
allowed_models=tuple(budget.allowed_models),
)
def _limits_audit_value(rows: "Sequence[prisma_models.LiteLLM_TeamMembership]") -> str:
"""Serialize the members' limits for an audit-log value, dropping the limits they do not set."""
return safe_dumps(
_AuditedMemberBudgets(
team_member_budgets=tuple(_audited_member_budget(row) for row in sorted(rows, key=lambda row: row.user_id))
).model_dump(exclude_none=True, mode="json")
)
def _result(
member: TeamMemberBudgetPatch,
user_id: str | None,
@ -114,6 +172,8 @@ async def bulk_update_team_member_budgets(
user_api_key_dict: UserAPIKeyAuth,
prisma_client: PrismaClient,
user_api_key_cache: UserApiKeyCache,
litellm_proxy_admin_name: str,
litellm_changed_by: str | None = None,
) -> tuple[TeamMemberBudgetUpdateResult, ...]:
"""Apply one merge patch of per-member limits per requested member, in one transaction."""
team: Final = await TeamRepository(WriterPinnedClient(prisma_client.db)).find_by_id(team_id)
@ -151,7 +211,7 @@ async def bulk_update_team_member_budgets(
team_members_filter: Final = _team_users_filter(team_id, user_ids)
async with prisma_client.tx(timeout=_BATCH_TX_TIMEOUT) as tx:
memberships: Final = await _membership_tx_db(tx).find_many(where=team_members_filter)
memberships: Final = await _membership_tx_db(tx).find_many(where=team_members_filter, include=_WITH_BUDGET)
budget_id_of: Final = MappingProxyType({m.user_id: m.budget_id for m in memberships})
shared: Final = await _shared_budget_ids(
tx, frozenset(budget_id for budget_id in budget_id_of.values() if budget_id is not None)
@ -179,6 +239,17 @@ async def bulk_update_team_member_budgets(
user_id=user_id, team_id=team_id, user_api_key_cache=user_api_key_cache
)
await create_object_audit_log(
object_id=team_id,
action="updated",
litellm_changed_by=litellm_changed_by,
user_api_key_dict=user_api_key_dict,
litellm_proxy_admin_name=litellm_proxy_admin_name,
table_name=LitellmTableNames.TEAM_TABLE_NAME,
before_value=_limits_audit_value(memberships),
after_value=_limits_audit_value(written),
)
budget_of: Final = MappingProxyType({m.user_id: m.litellm_budget_table for m in written})
return tuple(
_result(

View file

@ -525,6 +525,42 @@ async def mistral_proxy_route(
return received_value
@router.api_route(
"/typesafe/{endpoint:path}",
methods=["GET", "POST"], # mutable-ok: FastAPI route metadata requires a list
tags=["TypeSafe AI Pass-through", "pass-through"], # mutable-ok: FastAPI route metadata requires a list
)
async def typesafe_proxy_route(
endpoint: str,
request: Request,
fastapi_response: Response,
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
):
"""[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)"""
base_target_url: Final = get_secret_str("TYPESAFE_API_BASE") or "https://api.typesafe.ai"
encoded_endpoint: Final = httpx.URL(endpoint).path
normalized_endpoint: Final = encoded_endpoint if encoded_endpoint.startswith("/") else f"/{encoded_endpoint}"
base_url: Final = httpx.URL(base_target_url)
updated_url: Final = base_url.copy_with(
path=HttpPassThroughEndpointHelpers.join_base_and_endpoint_path(base_url, normalized_endpoint),
)
typesafe_api_key: Final = passthrough_endpoint_router.get_credentials(
custom_llm_provider="typesafe",
region_name=None,
)
endpoint_func: Final = create_pass_through_route(
endpoint=endpoint,
target=str(updated_url),
custom_headers={ # mutable-ok: pass-through request headers require a mutable mapping
"Authorization": f"Bearer {typesafe_api_key}",
"Content-Type": "application/json",
},
custom_llm_provider="typesafe",
is_streaming_request=False,
)
return await endpoint_func(request, fastapi_response, user_api_key_dict)
@router.api_route(
"/milvus/{endpoint:path}",
methods=["GET", "POST", "PUT", "DELETE", "PATCH"],

View file

@ -0,0 +1,117 @@
from collections.abc import Mapping
from datetime import datetime
from typing import Final
import httpx
from pydantic import BaseModel, TypeAdapter, ValidationError
import litellm
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.litellm_core_utils.litellm_logging import (
get_standard_logging_object_payload, # pyright: ignore[reportUnknownVariableType] # legacy helper has an untyped signature
)
from litellm.proxy._types import PassThroughEndpointLoggingTypedDict
from litellm.types.utils import ModelResponse, StandardPassThroughResponseObject, Usage
class _TypeSafeUsage(BaseModel):
input_tokens: int = 0
output_tokens: int = 0
class _TypeSafeResponse(BaseModel):
model: str | None = None
usage: _TypeSafeUsage | None = None
class _RegistryPricing(BaseModel):
input_cost_per_token: float = 0.0
output_cost_per_token: float = 0.0
_TYPESAFE_RESPONSE_ADAPTER: Final = TypeAdapter(_TypeSafeResponse)
_REGISTRY_PRICING_ADAPTER: Final = TypeAdapter(_RegistryPricing)
def _parse_typesafe_response(response_body: Mapping[str, object]) -> _TypeSafeResponse:
try:
return _TYPESAFE_RESPONSE_ADAPTER.validate_python(response_body)
except ValidationError:
return _TypeSafeResponse()
def _pricing_for(model_keys: tuple[str, ...]) -> _RegistryPricing:
for model_key in model_keys:
if model_key not in litellm.model_cost: # pyright: ignore[reportUnknownMemberType] # registry is dynamically typed
continue
try:
return _REGISTRY_PRICING_ADAPTER.validate_python(
litellm.model_cost[model_key] # pyright: ignore[reportUnknownMemberType] # registry is dynamically typed
)
except ValidationError:
continue
return _RegistryPricing()
class TypeSafePassthroughLoggingHandler:
@staticmethod
def typesafe_passthrough_handler(
httpx_response: httpx.Response,
response_body: Mapping[str, object],
logging_obj: LiteLLMLoggingObj,
url_route: str,
result: str,
start_time: datetime,
end_time: datetime,
cache_hit: bool,
request_body: Mapping[str, object],
**kwargs: object,
) -> PassThroughEndpointLoggingTypedDict:
response: Final = _parse_typesafe_response(response_body)
response_model: Final = response.model
request_model_value: Final = request_body.get("model")
request_model: Final = request_model_value if isinstance(request_model_value, str) else None
logged_model: Final = response_model or request_model or "unknown"
model_name: Final = f"typesafe/{logged_model}"
usage: Final = response.usage or _TypeSafeUsage()
input_tokens: Final = usage.input_tokens
output_tokens: Final = usage.output_tokens
candidate_model_keys: Final = tuple(
f"typesafe/{model}" for model in (response_model, request_model) if model is not None
)
pricing: Final = _pricing_for(candidate_model_keys)
response_cost: Final = (
input_tokens * pricing.input_cost_per_token + output_tokens * pricing.output_cost_per_token
)
usage_object: Final = Usage(
prompt_tokens=input_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + output_tokens,
)
updated_kwargs: Final = { # mutable-ok: pass-through logging contract requires mutable kwargs
**kwargs,
"model": model_name,
"custom_llm_provider": "typesafe",
"response_cost": response_cost,
"combined_usage_object": usage_object,
}
logging_obj.model_call_details.update(
model=model_name,
custom_llm_provider="typesafe",
response_cost=response_cost,
)
standard_logging_object: Final = get_standard_logging_object_payload(
kwargs=updated_kwargs,
init_response_obj=ModelResponse(model=model_name, usage=usage_object),
start_time=start_time,
end_time=end_time,
logging_obj=logging_obj,
status="success",
)
return { # mutable-ok: pass-through logging contract requires mutable result
"result": StandardPassThroughResponseObject(response=result),
"kwargs": { # mutable-ok: pass-through logging contract requires mutable kwargs
**updated_kwargs,
"standard_logging_object": standard_logging_object,
},
}

View file

@ -1,5 +1,6 @@
import json
from datetime import datetime
from types import MappingProxyType
from typing import Any, Final
from urllib.parse import urlparse
@ -256,6 +257,25 @@ class PassThroughEndpointLogging:
)
standard_logging_response_object = comprehend_medical_handler_result["result"] # rebind-ok: elif-chain
kwargs = comprehend_medical_handler_result["kwargs"] # rebind-ok: elif-chain contract
elif self.is_typesafe_route(custom_llm_provider):
from .llm_provider_handlers.typesafe_passthrough_logging_handler import (
TypeSafePassthroughLoggingHandler,
)
typesafe_handler_result: Final = TypeSafePassthroughLoggingHandler.typesafe_passthrough_handler(
httpx_response=httpx_response,
response_body=response_body if isinstance(response_body, dict) else MappingProxyType({}),
logging_obj=logging_obj,
url_route=url_route,
result=result,
start_time=start_time,
end_time=end_time,
cache_hit=cache_hit,
request_body=request_body,
**kwargs,
)
standard_logging_response_object = typesafe_handler_result["result"]
kwargs = typesafe_handler_result["kwargs"]
elif self.is_vertex_ai_live_route(url_route):
from .llm_provider_handlers.vertex_ai_live_passthrough_logging_handler import (
VertexAILivePassthroughLoggingHandler,
@ -389,6 +409,9 @@ class PassThroughEndpointLogging:
def is_comprehend_medical_route(self, custom_llm_provider: str | None) -> bool:
return custom_llm_provider == "comprehendmedical"
def is_typesafe_route(self, custom_llm_provider: str | None) -> bool:
return custom_llm_provider == "typesafe"
def is_langfuse_route(self, url_route: str):
parsed_url: Final = urlparse(url_route)
for route in self.TRACKED_LANGFUSE_ROUTES:

View file

@ -348,6 +348,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
"audio_transcription",
"audio_speech",
"responses",
"evaluation",
"ocr",
"realtime",
]
@ -3754,6 +3755,8 @@ agentic_loop_internal_litellm_params: Final = [
# the provider.
TRUSTED_CALLBACK_VARS_FIELD: Final = "litellm_trusted_callback_vars"
ADDRESSED_RESPONSE_ID_FIELD: Final = "_litellm_addressed_response_id"
# Bedrock managed-batch deployment config, read from litellm_params by the batch and
# files transformations. Listed for the same reason as the fields above: these sit on
# a deployment that also serves chat, so leaking them into extra_body makes Bedrock
@ -3768,7 +3771,7 @@ bedrock_batch_litellm_params: Final = (
all_litellm_params = (
agentic_loop_internal_litellm_params
+ [TRUSTED_CALLBACK_VARS_FIELD, *bedrock_batch_litellm_params]
+ [TRUSTED_CALLBACK_VARS_FIELD, ADDRESSED_RESPONSE_ID_FIELD, *bedrock_batch_litellm_params]
+ [
"metadata",
"litellm_metadata",

View file

@ -69184,6 +69184,27 @@
"supports_reasoning": true,
"supports_vision": true
},
"typesafe/jev-1.13.0": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-latest": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-preview": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"wandb/zai-org/GLM-5.3-Flash": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_token": 1.5e-07,

View file

@ -427,6 +427,7 @@
"chat",
"completion",
"embedding",
"evaluation",
"guardrail",
"image_edit",
"image_generation",

View file

@ -35,8 +35,8 @@ def test_is_over_limit():
def test_auto_router_capability_limit() -> None:
"""Only the signed license's auto_router feature lifts the one-router limit; an API-verified
license (no airgapped data) and an airgapped license without the feature keep it."""
"""The signed license's auto_router feature or its "*" wildcard lifts the one-router limit; an
API-verified license (no airgapped data) and an airgapped license without either keep it."""
license_check = LicenseCheck()
license_check.airgapped_license_data = {"expiration_date": "2999-01-01", "allowed_features": ["auto_router"]}
assert license_check.auto_router_capability_limit() is None
@ -47,9 +47,18 @@ def test_auto_router_capability_limit() -> None:
}
assert license_check.auto_router_capability_limit() is None
license_check.airgapped_license_data = {"expiration_date": "2999-01-01", "allowed_features": ["*"]}
assert license_check.auto_router_capability_limit() is None
license_check.airgapped_license_data = {"expiration_date": "2999-01-01", "allowed_features": ["sso", "*"]}
assert license_check.auto_router_capability_limit() is None
license_check.airgapped_license_data = {"expiration_date": "2999-01-01", "allowed_features": ["sso"]}
assert license_check.auto_router_capability_limit() == 1
license_check.airgapped_license_data = {"expiration_date": "2999-01-01", "allowed_features": "*"}
assert license_check.auto_router_capability_limit() is None
license_check.airgapped_license_data = {"expiration_date": "2999-01-01"}
assert license_check.auto_router_capability_limit() == 1
@ -57,7 +66,9 @@ def test_auto_router_capability_limit() -> None:
assert license_check.auto_router_capability_limit() == 1
def _signed_license(expiration_date: str) -> tuple[RSAPublicKey, str]:
def _signed_license(
expiration_date: str, allowed_features: tuple[str, ...] = ("auto_router",)
) -> tuple[RSAPublicKey, str]:
import base64
from cryptography.hazmat.primitives import hashes
@ -65,7 +76,7 @@ def _signed_license(expiration_date: str) -> tuple[RSAPublicKey, str]:
private_key = rsa.generate_private_key(public_exponent=65537, key_size=2048)
message = json.dumps(
{"expiration_date": expiration_date, "user_id": "u", "allowed_features": ["auto_router"]}
{"expiration_date": expiration_date, "user_id": "u", "allowed_features": list(allowed_features)}
).encode()
signature = private_key.sign(
message,
@ -99,3 +110,19 @@ def test_valid_signed_license_with_auto_router_lifts_the_limit() -> None:
assert license_check.verify_license_without_api_request(public_key=public_key, license_key=license_key) is True
assert license_check.auto_router_capability_limit() is None
def test_valid_signed_wildcard_license_lifts_the_limit() -> None:
"""The license generator defaults allowed_features to ["*"], meaning every feature, so a wildcard
license grants auto_router the same way a license that names it does."""
license_check = LicenseCheck()
public_key, license_key = _signed_license("2999-01-01", allowed_features=("*",))
assert license_check.verify_license_without_api_request(public_key=public_key, license_key=license_key) is True
assert license_check.grants_feature("auto_router") is True
assert license_check.auto_router_capability_limit() is None
named_public_key, named_key = _signed_license("2999-01-01", allowed_features=("sso", "audit_logs"))
assert license_check.verify_license_without_api_request(public_key=named_public_key, license_key=named_key) is True
assert license_check.grants_feature("auto_router") is False
assert license_check.auto_router_capability_limit() == 1

View file

@ -3,13 +3,14 @@ import socket
import stat
from typing import Optional
import pytest
import yaml
from click.testing import CliRunner
from litellm.proxy.client.cli.commands.claude_settings import ClaudeSettingsError
from litellm.proxy.client.cli.commands.autoroute import commands as commands_module
from litellm.proxy.client.cli.commands.autoroute import process as process_module
from litellm.proxy.client.cli.commands.autoroute.commands import down, up
from litellm.proxy.client.cli.commands.autoroute.commands import autoroute_group, start, stop
from litellm.proxy.client.cli.commands.autoroute.process import PidRecord, ProcessLaunchError, write_pid_record
from litellm.proxy.client.cli.commands.up import BackupRecord as ClaudeBackupRecord
from litellm.proxy.client.cli.commands.up import write_backup
@ -46,14 +47,14 @@ def _silence_signal_handling(monkeypatch):
monkeypatch.setattr(commands_module, "stream_log", lambda *a, **k: None)
class TestUpCommand:
class TestStartCommand:
def setup_method(self):
self.runner = CliRunner()
def test_refuses_when_never_configured(self, monkeypatch, tmp_path):
_patch_paths(monkeypatch, tmp_path)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code != 0
assert "lite autoroute configure" in result.output
@ -66,14 +67,14 @@ class TestUpCommand:
config_path.write_text("")
monkeypatch.setattr(commands_module, "is_port_available", lambda port: True)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code != 0
assert result.exception is None or isinstance(result.exception, SystemExit)
assert "lite autoroute configure" in result.output
def test_refuses_with_actionable_error_when_proxy_runtime_missing(self, monkeypatch, tmp_path):
"""`up` launches a real litellm proxy, which the thin `litellm[cli]` install cannot run.
"""`start` launches a real litellm proxy, which the thin `litellm[cli]` install cannot run.
It must fail fast with an actionable message pointing at the proxy install, before it ever
tries to launch the doomed subprocess (which would otherwise die with a bare ImportError)."""
config_path, _log_path, _settings_path, _backup_path, _pid_record_path = _patch_paths(monkeypatch, tmp_path)
@ -85,7 +86,7 @@ class TestUpCommand:
monkeypatch.setattr(commands_module, "launch_proxy", _fail_if_launched)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code != 0
assert "fastapi, websockets" in result.output
@ -99,18 +100,18 @@ class TestUpCommand:
)
monkeypatch.setattr(commands_module, "is_running", lambda pid: True)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code != 0
assert "already running" in result.output
assert "lite autoroute down" in result.output
assert "lite autoroute stop" in result.output
assert config_path.read_text() == yaml.safe_dump({"model_list": []})
def test_refuses_when_backup_exists_after_an_unclean_crash(self, monkeypatch, tmp_path):
"""A prior `up` that was SIGKILL'd leaves no live pid but does leave a stale backup file.
"""A prior `start` that was SIGKILL'd leaves no live pid but does leave a stale backup file.
Without this guard, a fresh `up` would overwrite that backup with the currently-patched
(not original) Claude settings, so `down`/Ctrl-C would restore the wrong content forever.
Without this guard, a fresh `start` would overwrite that backup with the currently-patched
(not original) Claude settings, so `stop`/Ctrl-C would restore the wrong content forever.
"""
config_path, _log_path, claude_settings_path, backup_path, _pid_record_path = _patch_paths(
monkeypatch, tmp_path
@ -119,11 +120,11 @@ class TestUpCommand:
claude_settings_path.write_text(json.dumps({"env": {"ANTHROPIC_AUTH_TOKEN": "stale-patched-token"}}))
write_backup(ClaudeBackupRecord(existed=True, content={"theme": "dark"}), backup_path)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code != 0
assert "already exists" in result.output
assert "lite autoroute down" in result.output
assert "lite autoroute stop" in result.output
assert json.loads(backup_path.read_text())["content"] == {"theme": "dark"}
def test_happy_path_patches_settings_then_restores_everything_on_stop(self, monkeypatch, tmp_path):
@ -151,7 +152,7 @@ class TestUpCommand:
monkeypatch.setattr("threading.Event.wait", fake_wait)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code == 0, result.output
assert captured["backup_existed"] is True
@ -198,7 +199,7 @@ class TestUpCommand:
monkeypatch.setattr("threading.Event.wait", fake_wait)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code == 0, result.output
assert "invalid or unexpected JSON" in result.output
@ -222,7 +223,7 @@ class TestUpCommand:
monkeypatch.setattr(commands_module, "terminate", lambda pid, **k: terminate_calls.append(pid))
monkeypatch.setattr(commands_module.secrets, "token_urlsafe", lambda n: "fixed-master-key")
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code != 0
assert "boom" in result.output
@ -234,7 +235,7 @@ class TestUpCommand:
def test_terminates_ephemeral_proxy_when_claude_settings_is_corrupt(self, monkeypatch, tmp_path):
"""The health check can pass and the proxy can come up fine, but if
~/.claude/settings.json turns out to be corrupt, the just-started proxy must not be left
running with no pid record -- exactly the leak `lite autoroute down` exists to clean up."""
running with no pid record -- exactly the leak `lite autoroute stop` exists to clean up."""
config_path, _log_path, claude_settings_path, backup_path, pid_record_path = _patch_paths(monkeypatch, tmp_path)
config_path.write_text(yaml.safe_dump({"model_list": []}))
claude_settings_path.write_text("not json at all {{{")
@ -247,7 +248,7 @@ class TestUpCommand:
monkeypatch.setattr(commands_module, "terminate", lambda pid, **k: terminate_calls.append(pid))
monkeypatch.setattr(commands_module.secrets, "token_urlsafe", lambda n: "fixed-master-key")
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code != 0
assert "invalid JSON" in result.output
@ -257,7 +258,7 @@ class TestUpCommand:
def test_a_status_line_install_failure_leaves_no_backup_behind(self, monkeypatch, tmp_path):
# The install runs before the backup is written, so a failure cannot strand a backup that
# would make every later `lite configure` / `lite autoroute up` think a session still owns settings.json
# would make every later `lite configure` / `lite autoroute start` think a session still owns settings.json
config_path, _log_path, claude_settings_path, backup_path, pid_record_path = _patch_paths(monkeypatch, tmp_path)
config_path.write_text(yaml.safe_dump({"model_list": []}))
claude_settings_path.write_text(json.dumps({"theme": "dark"}))
@ -274,7 +275,7 @@ class TestUpCommand:
monkeypatch.setattr(commands_module, "install_statusline_script", boom)
monkeypatch.setattr(commands_module.secrets, "token_urlsafe", lambda n: "fixed-master-key")
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code != 0 and "disk full" in result.output
assert terminate_calls == [778]
@ -282,7 +283,7 @@ class TestUpCommand:
assert not backup_path.exists()
assert json.loads(claude_settings_path.read_text()) == {"theme": "dark"}
def test_up_uses_the_same_port_and_master_key_across_runs(self, monkeypatch, tmp_path):
def test_start_uses_the_same_port_and_master_key_across_runs(self, monkeypatch, tmp_path):
"""The LIT-4607/LIT-4608 regression: a client configured against one session must keep
working in the next, so consecutive runs must patch settings with an identical base URL
and auth token, and the key must be minted exactly once."""
@ -315,9 +316,9 @@ class TestUpCommand:
monkeypatch.setattr("threading.Event.wait", fake_wait)
first = self.runner.invoke(up)
first = self.runner.invoke(start)
run_index["current"] = 1
second = self.runner.invoke(up)
second = self.runner.invoke(start)
assert first.exit_code == 0, first.output
assert second.exit_code == 0, second.output
@ -326,7 +327,7 @@ class TestUpCommand:
assert captured[0]["ANTHROPIC_AUTH_TOKEN"] == captured[1]["ANTHROPIC_AUTH_TOKEN"]
assert mint_calls == [32]
def test_up_reuses_a_master_key_already_persisted_in_the_config(self, monkeypatch, tmp_path):
def test_start_reuses_a_master_key_already_persisted_in_the_config(self, monkeypatch, tmp_path):
config_path, _log_path, claude_settings_path, _backup_path, _pid_record_path = _patch_paths(
monkeypatch, tmp_path
)
@ -354,13 +355,13 @@ class TestUpCommand:
monkeypatch.setattr("threading.Event.wait", fake_wait)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code == 0, result.output
assert captured["env"]["ANTHROPIC_AUTH_TOKEN"] == "persisted-key"
assert captured["config_text"] == original_config
def test_up_mints_a_fresh_key_when_the_persisted_master_key_is_blank(self, monkeypatch, tmp_path):
def test_start_mints_a_fresh_key_when_the_persisted_master_key_is_blank(self, monkeypatch, tmp_path):
config_path, _log_path, claude_settings_path, _backup_path, _pid_record_path = _patch_paths(
monkeypatch, tmp_path
)
@ -382,16 +383,22 @@ class TestUpCommand:
monkeypatch.setattr("threading.Event.wait", fake_wait)
result = self.runner.invoke(up)
result = self.runner.invoke(start)
assert result.exit_code == 0, result.output
assert captured["env"]["ANTHROPIC_AUTH_TOKEN"] == "fresh-minted-key"
written_config = yaml.safe_load(config_path.read_text())
assert written_config["general_settings"]["master_key"] == "fresh-minted-key"
def test_port_override_reaches_settings_launch_and_pid_record(self, monkeypatch, tmp_path):
"""A --port override must flow to every consumer of the port; a hardcoded default in any
one of them would leave the patched settings pointing somewhere the proxy is not."""
@pytest.mark.parametrize(
("command", "leading_args"),
[(start, []), (autoroute_group, ["start"]), (autoroute_group, ["up"])],
ids=["start", "group start", "deprecated up alias"],
)
def test_port_override_reaches_settings_launch_and_pid_record(self, monkeypatch, tmp_path, command, leading_args):
"""A --port override must flow to every consumer of the port, through the deprecated `up`
alias too; a hardcoded default in any one of them would leave the patched settings pointing
somewhere the proxy is not."""
config_path, _log_path, claude_settings_path, _backup_path, pid_record_path = _patch_paths(
monkeypatch, tmp_path
)
@ -420,16 +427,16 @@ class TestUpCommand:
monkeypatch.setattr("threading.Event.wait", fake_wait)
result = self.runner.invoke(up, ["--port", "6111"])
result = self.runner.invoke(command, [*leading_args, "--port", "6111"])
assert result.exit_code == 0, result.output
assert captured["env"]["ANTHROPIC_BASE_URL"] == "http://127.0.0.1:6111"
assert launched_ports == [6111]
assert captured["pid_record"]["port"] == 6111
def test_up_rejects_port_4000_which_the_child_proxy_rebinds_unpredictably(self, monkeypatch, tmp_path):
def test_start_rejects_port_4000_which_the_child_proxy_rebinds_unpredictably(self, monkeypatch, tmp_path):
"""proxy_cli special-cases a busy port 4000 by silently rebinding to a random port,
which would desync base_url from the child; up must refuse 4000 outright."""
which would desync base_url from the child; start must refuse 4000 outright."""
config_path, _log_path, _settings_path, backup_path, _pid_record_path = _patch_paths(monkeypatch, tmp_path)
config_path.write_text(yaml.safe_dump({"model_list": []}))
@ -438,13 +445,13 @@ class TestUpCommand:
monkeypatch.setattr(commands_module, "launch_proxy", _fail_launch)
result = self.runner.invoke(up, ["--port", "4000"])
result = self.runner.invoke(start, ["--port", "4000"])
assert result.exit_code != 0
assert "4000" in result.output
assert not backup_path.exists()
def test_up_refuses_when_the_port_is_busy_without_touching_any_state(self, monkeypatch, tmp_path):
def test_start_refuses_when_the_port_is_busy_without_touching_any_state(self, monkeypatch, tmp_path):
"""A busy port must fail loudly before anything is minted, launched, or patched --
never silently move to another port (the pre-fix behavior this ticket removes)."""
config_path, _log_path, claude_settings_path, backup_path, _pid_record_path = _patch_paths(
@ -463,18 +470,18 @@ class TestUpCommand:
sock.bind(("127.0.0.1", 0))
sock.listen(1)
busy_port = sock.getsockname()[1]
result = self.runner.invoke(up, ["--port", str(busy_port)])
result = self.runner.invoke(start, ["--port", str(busy_port)])
assert result.exit_code != 0
assert str(busy_port) in result.output
assert "lite autoroute down" in result.output
assert "lite autoroute stop" in result.output
assert "--port" in result.output
assert config_path.read_text() == original_config
assert not backup_path.exists()
assert json.loads(claude_settings_path.read_text()) == {"theme": "dark"}
class TestDownCommand:
class TestStopCommand:
def setup_method(self):
self.runner = CliRunner()
@ -491,7 +498,7 @@ class TestDownCommand:
monkeypatch.setattr(commands_module, "is_running", lambda pid: True)
monkeypatch.setattr(commands_module, "terminate", lambda pid, **k: terminate_calls.append(pid))
result = self.runner.invoke(down)
result = self.runner.invoke(stop)
assert result.exit_code == 0, result.output
assert "Stopped leftover ephemeral proxy" in result.output
@ -501,19 +508,33 @@ class TestDownCommand:
assert not backup_path.exists()
assert json.loads(claude_settings_path.read_text()) == original_settings
def test_removes_settings_that_did_not_exist_before_start(self, monkeypatch, tmp_path):
_config_path, _log_path, claude_settings_path, backup_path, _pid_record_path = _patch_paths(
monkeypatch, tmp_path
)
write_backup(ClaudeBackupRecord(existed=False, content=None), backup_path)
claude_settings_path.write_text(json.dumps({"env": {"ANTHROPIC_AUTH_TOKEN": "fixed-master-key"}}))
result = self.runner.invoke(stop)
assert result.exit_code == 0, result.output
assert f"Removed {claude_settings_path} (it did not exist before `lite autoroute start`)." in result.output
assert not claude_settings_path.exists()
assert not backup_path.exists()
def test_is_a_clean_no_op_when_nothing_is_running_and_no_backup_exists(self, monkeypatch, tmp_path):
_config_path, _log_path, claude_settings_path, _backup_path, _pid_record_path = _patch_paths(
monkeypatch, tmp_path
)
result = self.runner.invoke(down)
result = self.runner.invoke(stop)
assert result.exit_code == 0, result.output
assert "Nothing to restore." in result.output
assert not claude_settings_path.exists()
def test_clears_a_corrupt_pid_record_and_still_restores_settings(self, monkeypatch, tmp_path):
"""down is specifically the crash-recovery path -- a pid file truncated by a mid-write
"""stop is specifically the crash-recovery path -- a pid file truncated by a mid-write
crash must not block it from clearing the record and restoring Claude settings anyway."""
_config_path, _log_path, claude_settings_path, backup_path, pid_record_path = _patch_paths(
monkeypatch, tmp_path
@ -524,7 +545,7 @@ class TestDownCommand:
write_backup(ClaudeBackupRecord(existed=True, content=original_settings), backup_path)
claude_settings_path.write_text(json.dumps({"env": {"ANTHROPIC_AUTH_TOKEN": "fixed-master-key"}}))
result = self.runner.invoke(down)
result = self.runner.invoke(stop)
assert result.exit_code == 0, result.output
assert "invalid or unexpected JSON" in result.output
@ -540,7 +561,48 @@ class TestDownCommand:
backup_path.parent.mkdir(parents=True, exist_ok=True)
backup_path.write_text("not json at all {{{")
result = self.runner.invoke(down)
result = self.runner.invoke(stop)
assert result.exit_code != 0
assert "invalid or unexpected JSON" in result.output
class TestSubcommandNames:
def test_start_and_stop_are_the_listed_commands(self):
"""`lite up` already routes an existing proxy into Claude Code, so the ephemeral proxy's
launcher and its recovery path are listed as `start` and `stop`; the old names stay callable
but are hidden from the listing."""
runner = CliRunner()
listing = runner.invoke(autoroute_group, ["--help"])
assert listing.exit_code == 0, listing.output
listed = {line.split()[0] for line in listing.output.splitlines() if line.startswith(" ")}
assert {"configure", "start", "stop"} <= listed
assert listed.isdisjoint({"up", "down"})
for name in ("start", "stop", "up", "down"):
result = runner.invoke(autoroute_group, [name, "--help"])
assert result.exit_code == 0, result.output
assert "Show this message and exit" in result.output
def test_up_warns_then_behaves_like_start(self, monkeypatch, tmp_path):
_patch_paths(monkeypatch, tmp_path)
runner = CliRunner()
result = runner.invoke(autoroute_group, ["up", "--port", "5555"])
assert result.exit_code == 1, result.output
assert "`lite autoroute up` is deprecated" in result.stderr
assert "run `lite autoroute start` instead" in result.stderr
assert "No config found. Run `lite autoroute configure` first." in result.output
def test_down_warns_then_behaves_like_stop(self, monkeypatch, tmp_path):
_patch_paths(monkeypatch, tmp_path)
runner = CliRunner()
result = runner.invoke(autoroute_group, ["down"])
assert result.exit_code == 0, result.output
assert "`lite autoroute down` is deprecated" in result.stderr
assert "run `lite autoroute stop` instead" in result.stderr
assert "Nothing to restore." in result.output

View file

@ -39,7 +39,7 @@ from litellm.proxy.client.cli.commands.claude_settings import (
def _owners(*backup_paths):
"""Stand-in owners for the real `lite up` / `lite autoroute up` registry."""
"""Stand-in owners for the real `lite up` / `lite autoroute start` registry."""
return tuple(SettingsFileOwner(path, "lite up", "lite down") for path in backup_paths)
@ -162,7 +162,7 @@ class TestConfigureClaudeSettings:
class TestConflictingOwnersOfTheSettingsFile:
"""Both `lite up` and `lite autoroute up` restore a backup when they stop.
"""Both `lite up` and `lite autoroute start` restore a backup when they stop.
Guarding only one of them leaves the other free to silently revert this
write, which is the exact hazard the guard exists to prevent.
@ -184,11 +184,11 @@ class TestConflictingOwnersOfTheSettingsFile:
settings_path = tmp_path / "claude" / "settings.json"
backup = tmp_path / "auto.json"
backup.write_text("{}")
autoroute = SettingsFileOwner(backup, "lite autoroute up", "lite autoroute down")
autoroute = SettingsFileOwner(backup, "lite autoroute start", "lite autoroute stop")
with pytest.raises(ClaudeSettingsError, match="`lite autoroute up` is currently managing"):
with pytest.raises(ClaudeSettingsError, match="`lite autoroute start` is currently managing"):
_static_configure("https://proxy.example.com", settings_path, (autoroute,))
with pytest.raises(ClaudeSettingsError, match="Run `lite autoroute down` first"):
with pytest.raises(ClaudeSettingsError, match="Run `lite autoroute stop` first"):
_static_configure("https://proxy.example.com", settings_path, (autoroute,))
def test_the_registry_matches_the_paths_the_commands_actually_use(self):
@ -197,7 +197,7 @@ class TestConflictingOwnersOfTheSettingsFile:
assert AUTOROUTE_BACKUP_PATH == AUTOROUTE_DIR / "claude_settings_backup.json"
assert {o.backup_path for o in SETTINGS_FILE_OWNERS} == {BACKUP_PATH, AUTOROUTE_BACKUP_PATH}
assert {o.stop_command for o in SETTINGS_FILE_OWNERS} == {"lite down", "lite autoroute down"}
assert {o.stop_command for o in SETTINGS_FILE_OWNERS} == {"lite down", "lite autoroute stop"}
class TestDoesNotDestroyUserOwnedStructure:
@ -297,7 +297,7 @@ class TestConfigureStatePath:
class TestMergeClaudeSettings:
"""One merge for every way Claude Code gets wired: `lite up`, `lite configure claude` and `lite autoroute up`."""
"""One merge for every way Claude Code gets wired: `lite up`, `lite configure claude` and `lite autoroute start`."""
def test_a_static_token_lands_in_env_and_the_helper_slot_is_cleared(self):
settings = {"apiKeyHelper": "/usr/local/bin/lite auth print-token", "env": {"ANTHROPIC_API_KEY": "leaked"}}
@ -337,7 +337,7 @@ class TestMergeClaudeSettings:
def test_a_tier_model_forces_every_claude_code_tier_as_autoroute_needs(self):
# Router's auto-router registry is keyed by the literal requested model string with no
# wildcard resolution, so `lite autoroute up` overrides the env var each tier reads.
# wildcard resolution, so `lite autoroute start` overrides the env var each tier reads.
settings = {"env": {"ANTHROPIC_DEFAULT_SONNET_MODEL": "claude-opus-4-8"}}
merged = merge_claude_settings(
settings, "http://127.0.0.1:4000", StaticToken("token-abc"), tier_model="autorouter"

View file

@ -1,6 +1,6 @@
# tests/litellm/proxy/common_utils/test_upsert_budget_membership.py
import types
from datetime import datetime, timezone
from datetime import datetime, timedelta, timezone
from unittest.mock import AsyncMock, MagicMock
import pytest
@ -27,9 +27,7 @@ def mock_tx():
budget = MagicMock()
budget.update = AsyncMock()
budget.find_unique = AsyncMock(return_value=None)
budget.create = AsyncMock(
return_value=types.SimpleNamespace(budget_id="new-budget-123")
)
budget.create = AsyncMock(return_value=types.SimpleNamespace(budget_id="new-budget-123"))
tx = MagicMock()
tx.litellm_teammembership = membership
@ -83,9 +81,7 @@ async def test_empty_patch_is_noop(mock_tx, fake_user):
# member falls back to the team default instead of keeping an empty private row.
@pytest.mark.asyncio
async def test_clearing_all_limits_disconnects(mock_tx, fake_user):
mock_tx.litellm_budgettable.find_unique = AsyncMock(
return_value=budget_row(max_budget=100.0)
)
mock_tx.litellm_budgettable.find_unique = AsyncMock(return_value=budget_row(max_budget=100.0))
await _upsert_budget_and_membership(
mock_tx,
@ -136,9 +132,7 @@ async def test_clear_one_field_keeps_others(mock_tx, fake_user):
# budget_reset_at, so the budget rolls over without waiting for the reset cron.
@pytest.mark.asyncio
async def test_update_in_place_seeds_reset_at(mock_tx, fake_user):
mock_tx.litellm_budgettable.find_unique = AsyncMock(
return_value=budget_row(max_budget=20.0)
)
mock_tx.litellm_budgettable.find_unique = AsyncMock(return_value=budget_row(max_budget=20.0))
await _upsert_budget_and_membership(
mock_tx,
@ -163,9 +157,7 @@ async def test_update_in_place_seeds_reset_at(mock_tx, fake_user):
# budget_duration must not get a (re)computed reset time.
@pytest.mark.asyncio
async def test_update_in_place_single_field_leaves_reset_at_alone(mock_tx, fake_user):
mock_tx.litellm_budgettable.find_unique = AsyncMock(
return_value=budget_row(max_budget=50.0)
)
mock_tx.litellm_budgettable.find_unique = AsyncMock(return_value=budget_row(max_budget=50.0))
await _upsert_budget_and_membership(
mock_tx,
@ -294,6 +286,7 @@ async def test_create_from_temp_pair_skips_zero_team_default_cap(mock_tx, fake_u
@pytest.mark.asyncio
async def test_clone_on_write_from_shared_default(mock_tx, fake_user):
shared_default_id = "team-default-budget-1"
shared_reset_at = datetime.now(timezone.utc) + timedelta(hours=3)
mock_tx.litellm_budgettable.find_unique = AsyncMock(
return_value=budget_row(
budget_id=shared_default_id,
@ -304,6 +297,7 @@ async def test_clone_on_write_from_shared_default(mock_tx, fake_user):
rpm_limit=None,
model_max_budget=None,
budget_duration="1d",
budget_reset_at=shared_reset_at,
allowed_models=[],
)
)
@ -321,7 +315,7 @@ async def test_clone_on_write_from_shared_default(mock_tx, fake_user):
mock_tx.litellm_budgettable.update.assert_not_called()
mock_tx.litellm_budgettable.create.assert_awaited_once()
create_data = mock_tx.litellm_budgettable.create.await_args.kwargs["data"]
assert_future_reset_time(create_data.pop("budget_reset_at"))
assert create_data.pop("budget_reset_at") == shared_reset_at
assert create_data == {
"created_by": fake_user.user_id,
"updated_by": fake_user.user_id,
@ -387,9 +381,7 @@ async def test_clone_on_write_clears_duration(mock_tx, fake_user):
# team default), we update it in place rather than forking another row.
@pytest.mark.asyncio
async def test_private_budget_updates_in_place(mock_tx, fake_user):
mock_tx.litellm_budgettable.find_unique = AsyncMock(
return_value=budget_row(max_budget=10.0)
)
mock_tx.litellm_budgettable.find_unique = AsyncMock(return_value=budget_row(max_budget=10.0))
await _upsert_budget_and_membership(
mock_tx,

View file

@ -7,6 +7,7 @@ table and the membership/budget relation the bulk budget writer needs.
"""
import copy
import json
from collections.abc import Mapping, Sequence
from contextlib import asynccontextmanager
from datetime import datetime, timedelta, timezone
@ -244,6 +245,7 @@ def _budget(
tpm_limit: int | None = None,
rpm_limit: int | None = None,
budget_duration: str | None = None,
budget_reset_at: datetime | None = None,
) -> _BudgetRow:
return _BudgetRow(
budget_id=budget_id,
@ -251,6 +253,7 @@ def _budget(
tpm_limit=tpm_limit,
rpm_limit=rpm_limit,
budget_duration=budget_duration,
budget_reset_at=budget_reset_at,
)
@ -267,6 +270,7 @@ async def _bulk_update(
user_api_key_dict=caller,
prisma_client=prisma, # pyright: ignore[reportArgumentType] # fake stands in for PrismaClient
user_api_key_cache=cache or UserApiKeyCache(),
litellm_proxy_admin_name="default_user_id",
)
@ -631,6 +635,95 @@ async def test_the_roster_authz_read_runs_on_the_writer_so_a_lagging_replica_can
assert writer.db.litellm_budgettable.rows["priv-m1"].max_budget == 1.0
@pytest.mark.asyncio
async def test_the_batch_writes_one_audit_entry_carrying_every_written_members_limits_before_and_after(monkeypatch):
import litellm
from litellm.proxy._types import LitellmTableNames
monkeypatch.setattr(litellm, "store_audit_logs", True)
captured: list[object] = []
async def capture(request_data):
captured.append(request_data)
monkeypatch.setattr("litellm.proxy.management_helpers.audit_logs.create_audit_log_for_update", capture)
prisma = _FakePrisma(
teams=[_team("m1", "m2")],
memberships=[_membership("m1", "priv-m1"), _membership("m2", "priv-m2")],
budgets=[_budget("priv-m1", max_budget=1.0), _budget("priv-m2", max_budget=2.0)],
)
await _bulk_update(prisma, [{"user_id": "m1", "max_budget_in_team": 10}])
assert len(captured) == 1
entry = captured[0]
assert (entry.object_id, entry.action, entry.table_name) == (
TEAM_ID,
"updated",
LitellmTableNames.TEAM_TABLE_NAME,
)
before = {row["user_id"]: row for row in json.loads(entry.before_value)["team_member_budgets"]}
after = {row["user_id"]: row for row in json.loads(entry.updated_values)["team_member_budgets"]}
assert (before["m1"]["max_budget"], after["m1"]["max_budget"]) == (1.0, 10.0)
assert "m2" not in before and "m2" not in after
@pytest.mark.asyncio
async def test_no_audit_entry_is_written_when_audit_logging_is_off(monkeypatch):
import litellm
monkeypatch.setattr(litellm, "store_audit_logs", False)
captured: list[object] = []
async def capture(request_data):
captured.append(request_data)
monkeypatch.setattr("litellm.proxy.management_helpers.audit_logs.create_audit_log_for_update", capture)
prisma = _FakePrisma(
teams=[_team("m1")],
memberships=[_membership("m1", "priv-m1")],
budgets=[_budget("priv-m1", max_budget=1.0)],
)
await _bulk_update(prisma, [{"user_id": "m1", "max_budget_in_team": 10}])
assert captured == []
assert _budget_of(prisma, "m1").max_budget == 10.0
@pytest.mark.asyncio
async def test_forking_a_shared_row_keeps_its_reset_window_so_an_unrelated_limit_edit_grants_no_free_period():
shared_reset_at = datetime.now(timezone.utc) + timedelta(days=3)
prisma = _FakePrisma(
teams=[_team("m1", "m2")],
memberships=[_membership("m1", "shared-b"), _membership("m2", "shared-b")],
budgets=[_budget("shared-b", max_budget=100.0, budget_duration="30d", budget_reset_at=shared_reset_at)],
)
results = await _bulk_update(prisma, [{"user_id": "m1", "tpm_limit": 9}])
assert [(r.success, r.budget_duration) for r in results] == [(True, "30d")]
assert _budget_id_of(prisma, "m1") not in (None, "shared-b")
assert _budget_of(prisma, "m1").budget_reset_at == shared_reset_at
assert prisma.db.litellm_budgettable.rows["shared-b"].budget_reset_at == shared_reset_at
@pytest.mark.asyncio
async def test_forking_a_shared_row_does_restart_the_window_when_the_patch_sets_a_new_duration():
shared_reset_at = datetime.now(timezone.utc) + timedelta(days=3)
prisma = _FakePrisma(
teams=[_team("m1", "m2")],
memberships=[_membership("m1", "shared-b"), _membership("m2", "shared-b")],
budgets=[_budget("shared-b", max_budget=100.0, budget_duration="30d", budget_reset_at=shared_reset_at)],
)
await _bulk_update(prisma, [{"user_id": "m1", "budget_duration": "1d"}])
forked = _budget_of(prisma, "m1").budget_reset_at
assert forked is not None and forked != shared_reset_at
assert forked <= datetime.now(timezone.utc) + timedelta(days=1)
app = FastAPI()

View file

@ -0,0 +1,134 @@
from datetime import datetime
from unittest.mock import MagicMock
import httpx
import pytest
import litellm
from litellm.proxy.pass_through_endpoints.llm_provider_handlers.typesafe_passthrough_logging_handler import (
TypeSafePassthroughLoggingHandler,
)
from litellm.proxy.pass_through_endpoints.success_handler import PassThroughEndpointLogging
@pytest.fixture(autouse=True)
def local_model_cost_map(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
def _response() -> httpx.Response:
return httpx.Response(
200,
request=httpx.Request("POST", "https://api.typesafe.ai/v1/systemone"),
json={"model": "jev-1.13.0"},
)
def _logging_obj() -> MagicMock:
logging_obj = MagicMock()
logging_obj.model_call_details = {}
return logging_obj
def _handler_result(response_body: dict, request_body: dict) -> dict:
return TypeSafePassthroughLoggingHandler.typesafe_passthrough_handler(
httpx_response=_response(),
response_body=response_body,
logging_obj=_logging_obj(),
url_route="https://api.typesafe.ai/v1/systemone",
result='{"answers": {}}',
start_time=datetime.now(),
end_time=datetime.now(),
cache_hit=False,
request_body=request_body,
)
def test_uses_registry_pricing_and_standard_usage():
logging_obj = _logging_obj()
model_key = "typesafe/jev-1.13.0"
model_cost = litellm.model_cost[model_key]
response = TypeSafePassthroughLoggingHandler.typesafe_passthrough_handler(
httpx_response=_response(),
response_body={"model": "jev-1.13.0", "usage": {"input_tokens": 312, "output_tokens": 48}},
logging_obj=logging_obj,
url_route="https://api.typesafe.ai/v1/systemone",
result='{"answers": {}}',
start_time=datetime.now(),
end_time=datetime.now(),
cache_hit=False,
request_body={"model": "jev-latest"},
)
expected_cost = 312 * model_cost["input_cost_per_token"] + 48 * model_cost["output_cost_per_token"]
assert response["kwargs"]["response_cost"] == pytest.approx(expected_cost)
assert response["kwargs"]["combined_usage_object"].prompt_tokens == 312
assert response["kwargs"]["combined_usage_object"].completion_tokens == 48
assert response["kwargs"]["combined_usage_object"].total_tokens == 360
def test_falls_back_to_request_model_when_response_model_is_missing():
result = _handler_result(
{"usage": {"input_tokens": 10, "output_tokens": 2}},
{"model": "jev-latest"},
)
model_cost = litellm.model_cost["typesafe/jev-latest"]
expected_cost = 10 * model_cost["input_cost_per_token"] + 2 * model_cost["output_cost_per_token"]
assert result["kwargs"]["model"] == "typesafe/jev-latest"
assert result["kwargs"]["response_cost"] == pytest.approx(expected_cost)
def test_call_naming_no_model_is_logged_as_unknown_and_never_priced_as_a_registry_model():
result = _handler_result({"usage": {"input_tokens": 10, "output_tokens": 2}}, {})
assert result["kwargs"]["model"] == "typesafe/unknown"
assert result["kwargs"]["response_cost"] == 0.0
def test_missing_usage_is_zero_cost():
result = _handler_result({"model": "jev-1.13.0"}, {"model": "jev-latest"})
assert result["kwargs"]["response_cost"] == 0.0
def test_records_model_provider_and_cost_on_logging_details():
logging_obj = _logging_obj()
result = TypeSafePassthroughLoggingHandler.typesafe_passthrough_handler(
httpx_response=_response(),
response_body={"model": "jev-1.13.0", "usage": {"input_tokens": 1, "output_tokens": 0}},
logging_obj=logging_obj,
url_route="https://api.typesafe.ai/v1/systemone",
result="{}",
start_time=datetime.now(),
end_time=datetime.now(),
cache_hit=False,
request_body={"model": "jev-latest"},
)
assert result["kwargs"]["model"] == "typesafe/jev-1.13.0"
assert result["kwargs"]["custom_llm_provider"] == "typesafe"
assert result["kwargs"]["response_cost"] > 0
assert logging_obj.model_call_details["model"] == "typesafe/jev-1.13.0"
assert logging_obj.model_call_details["custom_llm_provider"] == "typesafe"
assert logging_obj.model_call_details["response_cost"] == result["kwargs"]["response_cost"]
def test_success_handler_dispatches_to_typesafe_handler():
logging_obj = _logging_obj()
normalized = PassThroughEndpointLogging().normalize_llm_passthrough_logging_payload(
httpx_response=_response(),
response_body={"model": "jev-1.13.0", "usage": {"input_tokens": 1, "output_tokens": 0}},
request_body={"model": "jev-latest"},
logging_obj=logging_obj,
url_route="https://api.typesafe.ai/v1/systemone",
result="{}",
start_time=datetime.now(),
end_time=datetime.now(),
cache_hit=False,
custom_llm_provider="typesafe",
)
assert normalized["kwargs"]["custom_llm_provider"] == "typesafe"
assert normalized["kwargs"]["model"] == "typesafe/jev-1.13.0"

View file

@ -9,6 +9,7 @@ from types import MappingProxyType, SimpleNamespace
from typing import Final
from unittest import mock
from unittest.mock import AsyncMock, MagicMock, Mock, patch
from urllib.parse import parse_qs
import httpx
import pytest
@ -43,6 +44,7 @@ from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import (
mistral_proxy_route,
relay_nvidia_nim_request,
openai_proxy_route,
typesafe_proxy_route,
vertex_discovery_proxy_route,
vertex_proxy_route,
vllm_proxy_route,
@ -6136,3 +6138,51 @@ class TestAzureRelayDeploymentSegment:
)
assert [call["model"] for call in captured] == ["gpt", "gpt"]
class TestTypeSafePassthroughRoute:
@staticmethod
def _request(body: object, query_params: Mapping[str, str] | None = None) -> MagicMock:
request = MagicMock(spec=Request)
request.method = "POST"
request.query_params = query_params or {}
request.json = AsyncMock(return_value=body)
return request
@pytest.mark.asyncio
async def test_forwards_target_auth_headers_provider_and_query(self, monkeypatch):
monkeypatch.setenv("TYPESAFE_API_KEY", "typesafe-test-key")
monkeypatch.setenv("TYPESAFE_API_BASE", "https://typesafe.example/base")
async def fake_upstream(request, *_args):
target: Final = create_route.call_args.kwargs["target"]
upstream_url: Final = httpx.URL(target).copy_merge_params(request.query_params)
return {"upstream_query": parse_qs(upstream_url.query.decode())}
endpoint_func = AsyncMock(side_effect=fake_upstream)
create_route = Mock(return_value=endpoint_func)
monkeypatch.setattr(
"litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints.create_pass_through_route",
create_route,
)
request = self._request({"state": "x"}, {"trace": "yes"})
result = await typesafe_proxy_route(
endpoint="v1/systemone",
request=request,
fastapi_response=MagicMock(spec=Response),
user_api_key_dict=UserAPIKeyAuth(api_key="virtual-key"),
)
assert result == {"upstream_query": {"trace": ["yes"]}}
endpoint_func.assert_awaited_once()
create_route.assert_called_once_with(
endpoint="v1/systemone",
target="https://typesafe.example/base/v1/systemone",
custom_headers={
"Authorization": "Bearer typesafe-test-key",
"Content-Type": "application/json",
},
custom_llm_provider="typesafe",
is_streaming_request=False,
)

View file

@ -12,14 +12,21 @@ capture the forwarded kwargs; if the flag-setting line is removed the captured
kwargs lack the flag and these tests fail.
"""
import json
from collections.abc import Mapping
from typing import Final
from unittest.mock import patch
import httpx
import pytest
import litellm
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.responses.litellm_completion_transformation.handler import (
LiteLLMCompletionTransformationHandler,
)
from litellm.types.llms.openai import ResponsesAPIResponse
from litellm.types.utils import ADDRESSED_RESPONSE_ID_FIELD
class _StopForwarding(Exception):
@ -170,3 +177,49 @@ async def test_async_fallback_returns_hoisted_nested_custom_tool_call_as_custom_
tool_calls = [(item.type, item.name, item.input) for item in response.output if item.type == "custom_tool_call"]
assert tool_calls == [("custom_tool_call", "exec", "ls")]
class _RecordingAnthropicHandler:
def __init__(self, reply: Mapping[str, object]) -> None:
self.reply: Final = reply
self.request_body: Mapping[str, object] | None = None
def __call__(self, request: httpx.Request) -> httpx.Response:
self.request_body = json.loads(request.content)
return httpx.Response(200, json=dict(self.reply), request=request)
_ANTHROPIC_MESSAGE_PAYLOAD: Final = {
"id": "msg_turn_two",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-6",
"content": [{"type": "text", "text": "14"}],
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 1},
}
@pytest.mark.asyncio
async def test_bridged_follow_up_turn_keeps_the_addressed_response_id_off_the_provider_body():
provider: Final = _RecordingAnthropicHandler(_ANTHROPIC_MESSAGE_PAYLOAD)
client: Final = AsyncHTTPHandler()
client.client = httpx.AsyncClient(transport=httpx.MockTransport(provider))
response = await litellm.aresponses(
model="azure_ai/claude-sonnet-4-6",
api_base="https://fake-foundry-resource.services.ai.azure.com",
api_key="fake-api-key",
input="Double it",
previous_response_id="resp_turn_one",
client=client,
**{ADDRESSED_RESPONSE_ID_FIELD: "resp_turn_one"},
)
assert provider.request_body is not None, "the bridged turn never reached the provider"
assert ADDRESSED_RESPONSE_ID_FIELD not in provider.request_body, (
f"the addressed response id reached the provider body: {sorted(provider.request_body)}"
)
assert isinstance(response, ResponsesAPIResponse)
assert [item.type for item in response.output] == ["message"]

View file

@ -0,0 +1,17 @@
import pytest
import litellm
@pytest.fixture(autouse=True)
def local_model_cost_map(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
def test_typesafe_models_share_pricing_and_provider_metadata():
entries = [litellm.model_cost[f"typesafe/{model}"] for model in ("jev-1.13.0", "jev-latest", "jev-preview")]
assert {entry["input_cost_per_token"] for entry in entries} == {entries[0]["input_cost_per_token"]}
assert {entry["output_cost_per_token"] for entry in entries} == {entries[0]["output_cost_per_token"]}
assert {entry["litellm_provider"] for entry in entries} == {"typesafe"}

View file

@ -46,6 +46,7 @@ from litellm.types.utils import (
PromptTokensDetailsWrapper,
StreamingChoices,
Usage,
ADDRESSED_RESPONSE_ID_FIELD,
all_litellm_params,
bedrock_batch_litellm_params,
)
@ -818,6 +819,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"container",
"image_edit",
"embedding",
"evaluation",
"guardrail",
"image_generation",
"video_generation",
@ -4787,6 +4789,20 @@ def test_get_litellm_params_keys_never_reach_the_provider():
)
def test_addressed_response_id_never_reaches_the_provider():
kwargs = {
"a_real_provider_specific_param": 1,
ADDRESSED_RESPONSE_ID_FIELD: "resp_addressed-by-the-client",
}
non_default = get_non_default_completion_params(kwargs)
assert non_default == {"a_real_provider_specific_param": 1}, (
"the addressed response id leaked into the provider params: "
f"{sorted(set(non_default) - {'a_real_provider_specific_param'})}"
)
def test_bedrock_batch_params_never_reach_the_provider():
"""A Bedrock managed-batch deployment carries aws_batch_role_arn / s3_* /
bedrock_tags in its litellm_params, and the same deployment also serves chat.

View file

@ -123,11 +123,11 @@ describe("CacheLeakageCard", () => {
expect(firstDataRow()).toHaveTextContent("alpha");
});
it("switches to the model view and lists only Anthropic models", () => {
it("switches to the model view and lists models from every provider", () => {
renderWith([
dayWithModels("2026-07-12", {
"claude-sonnet-5": { prompt_tokens: 5000, cache_read_input_tokens: 0 },
"gpt-4o": { prompt_tokens: 8000, cache_read_input_tokens: 0 },
"vertex_ai/gemini-2.5-pro": { prompt_tokens: 8000, cache_read_input_tokens: 2000 },
}),
]);
@ -135,7 +135,7 @@ describe("CacheLeakageCard", () => {
expect(screen.getByText("Cache leakage by model")).toBeInTheDocument();
expect(screen.getByText("claude-sonnet-5")).toBeInTheDocument();
expect(screen.queryByText("gpt-4o")).not.toBeInTheDocument();
expect(screen.getByText("vertex_ai/gemini-2.5-pro")).toBeInTheDocument();
});
it("shows an empty state when no key used tokens in the range", () => {

View file

@ -10,7 +10,6 @@ import {
classificationRatePer1kTurns,
computeCacheLeakage,
formatRangeLabel,
isAnthropicModel,
localIsoDay,
savingsSeriesOf,
toCumulative,
@ -209,20 +208,21 @@ describe("computeCacheLeakage", () => {
});
describe("computeCacheLeakage by model", () => {
it("aggregates only Anthropic models and ignores other providers", () => {
it("lists every provider's models, not only Anthropic", () => {
const models: Record<string, Partial<SpendMetrics>> = {
"claude-sonnet-5": { prompt_tokens: 10000, cache_read_input_tokens: 0 },
"anthropic/claude-haiku-4-5": { prompt_tokens: 4000, cache_read_input_tokens: 0 },
"bedrock/anthropic.claude-3-5-sonnet": { prompt_tokens: 2000, cache_read_input_tokens: 0 },
"gpt-4o": { prompt_tokens: 9000, cache_read_input_tokens: 0 },
"deepseek-chat": { prompt_tokens: 8000, cache_read_input_tokens: 0 },
"vertex_ai/gemini-2.5-pro": { prompt_tokens: 9000, cache_read_input_tokens: 3000 },
"bedrock/openai.gpt-5.6-luna": { prompt_tokens: 8000, cache_read_input_tokens: 0 },
"deepseek-chat": { prompt_tokens: 4000, cache_read_input_tokens: 0 },
};
const { rows } = computeCacheLeakage([modelDay("2026-07-01", models)], "model");
expect(rows.map((r) => r.id)).toEqual([
"claude-sonnet-5",
"anthropic/claude-haiku-4-5",
"bedrock/anthropic.claude-3-5-sonnet",
"bedrock/openai.gpt-5.6-luna",
"vertex_ai/gemini-2.5-pro",
"deepseek-chat",
]);
expect(rows.find((r) => r.id === "vertex_ai/gemini-2.5-pro")?.cacheHitRatio).toBeCloseTo(1 / 3, 6);
});
it("labels model rows by model name with no sublabel", () => {
@ -232,34 +232,20 @@ describe("computeCacheLeakage by model", () => {
expect(rows[0].sublabel).toBeNull();
});
it("prices model leakage at the Anthropic realized cache-read discount", () => {
it("prices model leakage at the realized cache-read discount across providers", () => {
const results = [
modelDay("2026-07-01", {
"claude-sonnet-5": { prompt_tokens: 1000, cache_read_input_tokens: 1000, prompt_caching_savings_spend: 2.0 },
"claude-haiku-4-5": { prompt_tokens: 500 },
"gemini-2.5-flash": { prompt_tokens: 500 },
}),
];
const { rows, netSavingsPerCachedToken } = computeCacheLeakage(results, "model");
expect(netSavingsPerCachedToken).toBeCloseTo(0.002, 6);
expect(rows.map((r) => r.id)).toEqual(["claude-haiku-4-5"]);
expect(rows.map((r) => r.id)).toEqual(["gemini-2.5-flash"]);
expect(rows[0].potentialSavings).toBeCloseTo(1.0, 6);
});
});
describe("isAnthropicModel", () => {
it("matches Claude-family models across providers and rejects others", () => {
const anthropic = [
"claude-sonnet-5",
"anthropic/claude-haiku-4-5",
"bedrock/anthropic.claude-3-5-sonnet",
"vertex_ai/claude-opus-4-8",
];
const others = ["gpt-4o", "deepseek-chat", "gemini-2.5-pro", "mistral-large"];
expect(anthropic.every(isAnthropicModel)).toBe(true);
expect(others.some(isAnthropicModel)).toBe(false);
});
});
describe("buildDailyToolSeries", () => {
const daily: ToolSpendDailyEntry[] = [
{ date: "2026-07-01", tool_name: "search", spend: 1.0, call_count: 1 },

View file

@ -44,8 +44,6 @@ export interface CacheLeakageResult {
netSavingsPerCachedToken: number | null;
}
export const isAnthropicModel = (model: string): boolean => /claude|anthropic/i.test(model);
interface LeakageAccumulator {
alias: string | null;
teamId: string | null;
@ -96,7 +94,6 @@ const aggregateByModel = (results: readonly DailyData[]): Map<string, LeakageAcc
const byModel = new Map<string, LeakageAccumulator>();
for (const day of results) {
for (const [model, entry] of Object.entries(day.breakdown?.models ?? {})) {
if (!isAnthropicModel(model)) continue;
const acc = byModel.get(model) ?? emptyAccumulator();
byModel.set(model, addMetrics(acc, entry.metrics, null, null));
}

View file

@ -16478,6 +16478,30 @@ export interface paths {
patch: operations["toolset_mcp_route_toolset__toolset_name__mcp_patch"];
trace?: never;
};
"/typesafe/{endpoint}": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* Typesafe Proxy Route
* @description [Docs](https://docs.litellm.ai/docs/pass_through/typesafe)
*/
get: operations["typesafe_proxy_route_typesafe__endpoint__get"];
put?: never;
/**
* Typesafe Proxy Route
* @description [Docs](https://docs.litellm.ai/docs/pass_through/typesafe)
*/
post: operations["typesafe_proxy_route_typesafe__endpoint__post"];
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/update/default_team_settings": {
parameters: {
query?: never;
@ -52222,7 +52246,10 @@ export interface operations {
bulk_update_team_member_budgets_action_management_v1_teams__team_id__members_bulk_update_post: {
parameters: {
query?: never;
header?: never;
header?: {
/** @description The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability */
"litellm-changed-by"?: string | null;
};
path: {
team_id: string;
};
@ -61618,6 +61645,68 @@ export interface operations {
};
};
};
typesafe_proxy_route_typesafe__endpoint__get: {
parameters: {
query?: never;
header?: never;
path: {
endpoint: string;
};
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": unknown;
};
};
/** @description Validation Error */
422: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["HTTPValidationError"];
};
};
};
};
typesafe_proxy_route_typesafe__endpoint__post: {
parameters: {
query?: never;
header?: never;
path: {
endpoint: string;
};
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": unknown;
};
};
/** @description Validation Error */
422: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["HTTPValidationError"];
};
};
};
};
update_default_team_settings_update_default_team_settings_patch: {
parameters: {
query?: never;