Merge remote-tracking branch 'origin/main' into litellm_auto-merge-drop-bot-gates

This commit is contained in:
kerry 2026-09-18 00:20:29 +00:00
commit 5d5f191a9c
137 changed files with 9986 additions and 1045 deletions

View file

@ -3,7 +3,7 @@
Example: Using CLI token with LiteLLM SDK
This example shows how to use the CLI authentication token
in your Python scripts after running `litellm-proxy login`.
in your Python scripts after running `lite login`.
"""
from textwrap import indent
@ -22,7 +22,7 @@ def main():
api_key = litellm.get_litellm_gateway_api_key()
if not api_key:
print("❌ No CLI token found. Please run 'litellm-proxy login' first.")
print("❌ No CLI token found. Please run 'lite login' first.")
return
print("✅ Found CLI token.")
@ -58,6 +58,6 @@ if __name__ == "__main__":
main()
print("\n💡 Tips:")
print("1. Run 'litellm-proxy login' to authenticate first")
print("1. Run 'lite login' to authenticate first")
print("2. Replace 'https://your-proxy.com' with your actual proxy URL")
print("3. The token is stored in your OS keychain, or in ~/.litellm/token.json when there is none")

View file

@ -1,614 +0,0 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "grafana",
"uid": "-- Grafana --"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"description": "",
"editable": true,
"fiscalYearStartMonth": 0,
"graphTooltip": 0,
"id": 2039,
"links": [],
"liveNow": false,
"panels": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "s"
},
"overrides": []
},
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 0
},
"id": 10,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "histogram_quantile(0.99, sum(rate(litellm_self_latency_bucket{self=\"self\"}[1m])) by (le))",
"legendFormat": "Time to first token",
"range": true,
"refId": "A"
}
],
"title": "Time to first token (latency)",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "currencyUSD"
},
"overrides": [
{
"matcher": {
"id": "byName",
"options": "7e4b0627fd32efdd2313c846325575808aadcf2839f0fde90723aab9ab73c78f"
},
"properties": [
{
"id": "displayName",
"value": "Translata"
}
]
}
]
},
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 8
},
"id": 11,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_spend_metric_total[30d])) by (hashed_api_key)",
"legendFormat": "{{team}}",
"range": true,
"refId": "A"
}
],
"title": "Spend by team",
"transformations": [],
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 9,
"w": 12,
"x": 0,
"y": 16
},
"id": 2,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum by (model) (increase(litellm_requests_metric_total[5m]))",
"legendFormat": "{{model}}",
"range": true,
"refId": "A"
}
],
"title": "Requests by model",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [],
"noValue": "0",
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 3,
"x": 0,
"y": 25
},
"id": 8,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"textMode": "auto"
},
"pluginVersion": "9.4.17",
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_llm_api_failed_requests_metric_total[1h]))",
"legendFormat": "__auto",
"range": true,
"refId": "A"
}
],
"title": "Faild Requests",
"type": "stat"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "currencyUSD"
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 3,
"x": 3,
"y": 25
},
"id": 6,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_spend_metric_total[30d])) by (model)",
"legendFormat": "{{model}}",
"range": true,
"refId": "A"
}
],
"title": "Spend",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 6,
"x": 6,
"y": 25
},
"id": 4,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_total_tokens_total[5m])) by (model)",
"legendFormat": "__auto",
"range": true,
"refId": "A"
}
],
"title": "Tokens",
"type": "timeseries"
}
],
"refresh": "1m",
"revision": 1,
"schemaVersion": 38,
"style": "dark",
"tags": [],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "prometheus",
"value": "edx8memhpd9tsa"
},
"hide": 0,
"includeAll": false,
"label": "datasource",
"multi": false,
"name": "DS_PROMETHEUS",
"options": [],
"query": "prometheus",
"queryValue": "",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
}
]
},
"time": {
"from": "now-1h",
"to": "now"
},
"timepicker": {},
"timezone": "",
"title": "LLM Proxy",
"uid": "rgRrHxESz",
"version": 15,
"weekStart": ""
}

View file

@ -1,6 +0,0 @@
## This folder contains the `json` for creating the following Grafana Dashboard
### Pre-Requisites
- Setup LiteLLM Proxy Prometheus Metrics https://docs.litellm.ai/docs/proxy/prometheus
![1716623265684](https://github.com/BerriAI/litellm/assets/29436595/0e12c57e-4a2d-4850-bd4f-e4294f87a814)

View file

@ -0,0 +1,11 @@
# LiteLLM All Prometheus Metrics dashboard
Every `litellm_*` metric family the proxy can expose on `/metrics` (134 families across 95 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
Import `grafana_dashboard.json` from **Dashboards > New > Import** and pick your Prometheus data source when prompted (the `DS_PROMETHEUS` variable). Counters are plotted as `rate()` over `$__rate_interval`, histograms as p50 / p95 / p99, gauges as the raw value grouped by the most useful label. Every query names the metric exactly as the proxy emits it (counters carry the `_total` suffix the Prometheus client adds), and `tests/test_litellm/integrations/test_prometheus_metric_name_consistency.py` fails if a metric is renamed without updating this dashboard
The first eleven rows need only `callbacks: ["prometheus"]`. The last three rows and the `litellm_admission_*` panels are emitted by other subsystems and stay empty until those are on: the service callback row needs `service_callback: ["prometheus_system"]` in `litellm_settings`, the circuit breaker row needs a Redis cache, the cleanup row needs spend log retention, and admission control needs its middleware enabled. Within the base rows, many panels only fill in once the matching feature is in use: budgets need keys, teams, users or orgs with `max_budget` set, cache panels need caching on, guardrail and MCP panels need those features configured, deployment health needs the router with more than one deployment or a failure to record, and `litellm_in_flight_requests` needs traffic at scrape time. An empty panel for a feature you do not use is expected
## Pre-requisites
Prometheus metrics on the proxy: https://docs.litellm.ai/docs/proxy/prometheus

View file

@ -476,7 +476,7 @@
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "topk(5, sort(litellm_remaining_requests))",
"expr": "topk(5, sort(litellm_remaining_requests_metric))",
"legendFormat": "__auto",
"range": true,
"refId": "A"
@ -573,7 +573,7 @@
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "topk(5, sort(litellm_remaining_tokens))",
"expr": "topk(5, sort(litellm_remaining_tokens_metric))",
"legendFormat": "__auto",
"range": true,
"refId": "A"

View file

@ -6,8 +6,14 @@ This folder contains the `json` for creating Grafana Dashboards
Charts the `gen_ai.*` metrics from the OpenTelemetry v2 integration: spend, tokens, request rate, and latency percentiles by model. Separate from the dashboards below, which chart the `litellm_*` Prometheus metrics.
## [LiteLLM All Prometheus Metrics dashboard](./dashboard_all_metrics)
Every `litellm_*` Prometheus metric family the proxy can emit (134 families, 95 panels) grouped by theme: traffic, latency, spend and tokens, cache, deployments, rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, plus the Redis circuit breaker, spend log cleanup and `prometheus_system` service metrics. Start here if you want everything on one screen; see its [readme](./dashboard_all_metrics/readme.md) for import steps and which panels need a feature enabled before they show data
## [LiteLLM v2 Dashboard](./dashboard_v2)
A compact view of proxy request rate, failures, latency and the top remaining-request / remaining-token gauges per model group
<img width="1316" alt="grafana_1" src="https://github.com/user-attachments/assets/d0df802d-0cb9-4906-a679-941c547789ab">
<img width="1289" alt="grafana_2" src="https://github.com/user-attachments/assets/b11f755f-e113-42ab-b21d-83f91f451a28">
<img width="1323" alt="grafana_3" src="https://github.com/user-attachments/assets/cb29ffdb-477d-4be1-a5cd-c3f7f2cb21c5">

View file

@ -96,6 +96,7 @@ GATEWAY_PATH_PREFIXES: tuple[str, ...] = (
"/langfuse/",
"/vllm/",
"/mistral/",
"/typesafe/",
"/nvidia_nim/",
"/groq/",
"/voyage/",

View file

@ -2016,6 +2016,7 @@ dependencies = [
"litellm-auth-azure",
"litellm-auth-gcp",
"litellm-framing",
"litellm-providers",
"mime_guess",
"moka",
"rand 0.8.7",
@ -2052,6 +2053,18 @@ dependencies = [
"tokio",
]
[[package]]
name = "litellm-providers"
version = "0.1.0"
dependencies = [
"litellm-auth",
"litellm-auth-aws",
"rstest",
"serde",
"serde_json",
"thiserror 2.0.19",
]
[[package]]
name = "litellm-python-bridge"
version = "0.1.0"

View file

@ -16,6 +16,7 @@ litellm-auth = { path = "crates/auth" }
litellm-auth-aws = { path = "crates/auth-aws" }
litellm-auth-azure = { path = "crates/auth-azure" }
litellm-auth-gcp = { path = "crates/auth-gcp" }
litellm-providers = { path = "crates/providers" }
litellm-cache = { path = "crates/cache" }
litellm-cache-memory = { path = "crates/cache-memory" }
litellm-token-counter = { path = "crates/token-counter" }

View file

@ -15,6 +15,7 @@ litellm-auth.workspace = true
litellm-auth-aws.workspace = true
litellm-auth-azure.workspace = true
litellm-auth-gcp.workspace = true
litellm-providers.workspace = true
litellm-framing.workspace = true
moka.workspace = true
mime_guess = "2.0.5"

View file

@ -24,3 +24,23 @@ pub enum Error {
#[error(transparent)]
Aws(#[from] litellm_auth_aws::Error),
}
impl From<litellm_providers::audio_transcription::Error> for Error {
fn from(error: litellm_providers::audio_transcription::Error) -> Self {
match error {
litellm_providers::audio_transcription::Error::InvalidType { expected, actual } => {
Self::InvalidType { expected, actual }
}
litellm_providers::audio_transcription::Error::MissingField(field) => {
Self::MissingField(field)
}
litellm_providers::audio_transcription::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::audio_transcription::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::audio_transcription::Error::Auth(error) => Self::Auth(error),
}
}
}

View file

@ -47,8 +47,8 @@ async fn signed_headers(
use std::collections::BTreeMap;
use std::time::SystemTime;
use crate::llms::base_llm::audio_transcription::transformation::AudioTranscriptionAuth;
use litellm_auth_aws::{aws_auth_config, resolve_credentials, sign_bedrock_post};
use litellm_providers::base_llm::audio_transcription::transformation::AudioTranscriptionAuth;
let AudioTranscriptionAuth::AwsSigV4 { region, .. } = &request.auth else {
return Ok(request.upstream_headers.clone());

View file

@ -3,7 +3,7 @@ pub use error::Error;
mod client;
mod handler;
mod prepare;
pub mod types;
pub use litellm_providers::audio_transcription::types;
pub use handler::execute_audio_transcription_provider_call;
pub use prepare::prepare_audio_transcription_provider_call;

View file

@ -4,10 +4,10 @@ use crate::http_utils::{has_header, string_headers};
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::audio_transcription::transformation::{
use litellm_providers::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
use crate::llms::bedrock::audio_transcription::BEDROCK_AUDIO_TRANSCRIPTION_CONFIG;
use litellm_providers::bedrock::audio_transcription::BEDROCK_AUDIO_TRANSCRIPTION_CONFIG;
fn provider_config(provider: &str) -> Option<&'static dyn BaseAudioTranscriptionConfig> {
if provider == "bedrock" {

View file

@ -2,8 +2,8 @@ use serde_json::{Map, Value};
use super::Error;
use crate::http_utils::string_headers as shared_string_headers;
use crate::llms::anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG;
use crate::llms::base_llm::chat::transformation::BaseConfig;
use litellm_providers::anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG;
use litellm_providers::base_llm::chat::transformation::BaseConfig;
const HEADER_CONTEXT: &str = "chat completions";
@ -11,7 +11,7 @@ pub(super) fn chat_completions_provider_config(provider: &str) -> Option<&'stati
match provider {
"anthropic" => Some(&ANTHROPIC_CHAT_COMPLETIONS_CONFIG),
"bedrock" => Some(
&crate::llms::bedrock::chat::converse_transformation::BEDROCK_CHAT_COMPLETIONS_CONFIG,
&litellm_providers::bedrock::chat::converse_transformation::BEDROCK_CHAT_COMPLETIONS_CONFIG,
),
_ => None,
}

View file

@ -24,3 +24,19 @@ pub enum Error {
#[error(transparent)]
Aws(#[from] litellm_auth_aws::Error),
}
impl From<litellm_providers::chat::Error> for Error {
fn from(error: litellm_providers::chat::Error) -> Self {
match error {
litellm_providers::chat::Error::MissingField(field) => Self::MissingField(field),
litellm_providers::chat::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::chat::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::chat::Error::Unsupported(reason) => Self::Unsupported(reason),
litellm_providers::chat::Error::Auth(error) => Self::Auth(error),
}
}
}

View file

@ -8,7 +8,7 @@ use super::types::{
ResolvedChatCompletionsRequest,
};
use crate::http_utils::{http_request, truncate_error_body};
use crate::llms::base_llm::chat::transformation::ChatCompletionsAuth;
use litellm_providers::base_llm::chat::transformation::ChatCompletionsAuth;
pub(super) async fn execute_chat_completions_provider_call(
request: ResolvedChatCompletionsRequest<'_>,
@ -59,6 +59,7 @@ pub(super) async fn execute_chat_completions_provider_call(
request
.config
.transform_response(&request.model, ProviderChatResponseData { body })
.map_err(Error::from)
.map_err(as_response_error)
}

View file

@ -10,12 +10,11 @@ mod error;
pub use error::Error;
mod client;
mod common_utils;
pub mod conversation;
pub use litellm_providers::chat::{conversation, response_utils};
pub(crate) mod handler;
mod prepare;
pub mod response_utils;
pub mod streaming;
pub mod types;
pub use litellm_providers::chat::types;
use handler::execute_chat_completions_provider_call;
use prepare::{parse_messages, resolve_provider_config, resolve_request};

View file

@ -10,7 +10,7 @@ use crate::http_utils::has_header;
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
use litellm_providers::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
pub(super) fn resolve_provider_config<'a>(
model: &'a str,

View file

@ -3,7 +3,7 @@ use serde_json::{Map, Value, json};
use super::Error;
use super::prepare::{prepare_provider_request, resolve_request};
use super::types::{ChatCompletionsRequest, ProviderChatCompletionsRequest};
use crate::llms::base_llm::chat::transformation::ChatCompletionsAuth;
use litellm_providers::base_llm::chat::transformation::ChatCompletionsAuth;
fn prepare_chat_completions_call(
request: ChatCompletionsRequest<'_>,

View file

@ -1,36 +1,4 @@
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct CustomLlmProvider<'a> {
pub model: &'a str,
pub custom_llm_provider: &'a str,
}
pub fn get_custom_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
) -> Option<CustomLlmProvider<'a>> {
if let Some(custom_llm_provider) = custom_llm_provider.filter(|provider| !provider.is_empty()) {
return Some(CustomLlmProvider {
model: strip_custom_llm_provider_prefix(model, custom_llm_provider),
custom_llm_provider,
});
}
let (custom_llm_provider, model) = model.split_once('/')?;
if custom_llm_provider.is_empty() || model.is_empty() {
return None;
}
Some(CustomLlmProvider {
model,
custom_llm_provider,
})
}
fn strip_custom_llm_provider_prefix<'a>(model: &'a str, custom_llm_provider: &str) -> &'a str {
model
.strip_prefix(custom_llm_provider)
.and_then(|model| model.strip_prefix('/'))
.unwrap_or(model)
}
pub use litellm_providers::provider_resolution::{CustomLlmProvider, get_custom_llm_provider};
#[cfg(test)]
mod tests {

View file

@ -1,2 +1 @@
pub mod streaming;
pub mod transformation;

View file

@ -2,16 +2,16 @@ use std::collections::HashMap;
use serde_json::Value;
use super::super::experimental_pass_through::messages::streaming::{
AnthropicContentBlock, AnthropicContentBlockDelta, AnthropicMessagesStreamEvent,
AnthropicStreamUsage,
};
use crate::chat_completions::Error;
use crate::chat_completions::streaming::StreamTransformer;
use crate::chat_completions::types::{
ChatCompletionChunk, ChatCompletionThinkingBlock, ChatCompletionToolCallChunk,
ChatCompletionsUsage,
};
use crate::llms::anthropic::experimental_pass_through::messages::streaming::{
AnthropicContentBlock, AnthropicContentBlockDelta, AnthropicMessagesStreamEvent,
AnthropicStreamUsage,
};
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum AnthropicJsonChunkType {

View file

@ -3,9 +3,9 @@ use serde_json::Value;
use time::OffsetDateTime;
use url::Url;
use crate::llms::anthropic::experimental_pass_through::messages::transformation::resolve_anthropic_api_base;
use crate::messages::Error;
use crate::messages::types::AnthropicMessagesResponse;
use litellm_providers::anthropic::experimental_pass_through::messages::transformation::resolve_anthropic_api_base;
const BATCHES_PATH_SUFFIX: &str = "/v1/messages/batches";

View file

@ -1,4 +1,3 @@
pub mod batches;
pub mod count_tokens;
pub mod streaming;
pub mod transformation;

View file

@ -1,2 +1 @@
pub mod anthropic;
pub(crate) mod ocr;

View file

@ -1,4 +1 @@
pub mod anthropic_messages;
pub mod audio_transcription;
pub mod chat;
pub(crate) mod ocr;

View file

@ -1,7 +1,6 @@
pub mod anthropic;
pub mod azure_ai;
pub mod base_llm;
pub mod bedrock;
pub(crate) mod cohere;
pub(crate) mod mistral;
pub mod openai;

View file

@ -3,9 +3,9 @@ use serde_json::{Map, Value};
use super::Error;
use crate::http_utils::string_headers as shared_string_headers;
pub(super) use crate::http_utils::{has_bearer_auth, has_header, truncate_error_body};
use crate::llms::anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG;
use crate::llms::azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG;
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use litellm_providers::anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG;
use litellm_providers::azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG;
use litellm_providers::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
const HEADER_CONTEXT: &str = "messages";

View file

@ -28,6 +28,22 @@ pub enum Error {
InvalidBedrockBase64(String),
}
impl From<litellm_providers::messages::Error> for Error {
fn from(error: litellm_providers::messages::Error) -> Self {
match error {
litellm_providers::messages::Error::MissingField(field) => Self::MissingField(field),
litellm_providers::messages::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::messages::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::messages::Error::Unsupported(reason) => Self::Unsupported(reason),
litellm_providers::messages::Error::Auth(error) => Self::Auth(error),
}
}
}
impl Error {
pub fn is_request(&self) -> bool {
match self {

View file

@ -40,6 +40,7 @@ pub(super) async fn execute_messages_provider_call(
request
.config
.transform_anthropic_messages_response(&request.model, response)
.map_err(Error::from)
}
pub(super) async fn execute_messages_provider_stream(

View file

@ -13,7 +13,7 @@ mod client;
mod common_utils;
mod handler;
mod prepare;
pub mod types;
pub use litellm_providers::messages::types;
use handler::{execute_messages_provider_call, execute_messages_provider_stream};
use types::{AnthropicMessagesResponse, MessagesRequest};

View file

@ -6,7 +6,7 @@ use super::types::{MessagesRequest, ProviderMessagesRequest};
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::anthropic_messages::transformation::{
use litellm_providers::base_llm::anthropic_messages::transformation::{
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
};

View file

@ -0,0 +1,16 @@
[package]
name = "litellm-providers"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
litellm-auth.workspace = true
litellm-auth-aws.workspace = true
serde.workspace = true
serde_json.workspace = true
thiserror.workspace = true
[dev-dependencies]
rstest.workspace = true

View file

@ -1,7 +1,7 @@
use serde_json::json;
use super::*;
use crate::chat_completions::Error;
use crate::chat::Error;
fn messages(value: Value) -> Vec<ChatMessage> {
serde_json::from_value(value).expect("valid messages")

View file

@ -1,19 +1,19 @@
use serde_json::{Map, Value, json};
use crate::chat_completions::Error;
use crate::chat_completions::conversation::{Conversation, build_conversation};
use crate::chat_completions::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat_completions::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
ProviderChatRequestData, ProviderChatResponseData,
};
use crate::constants::ANTHROPIC_OAUTH_TOKEN_PREFIX;
use crate::llms::anthropic::experimental_pass_through::messages::transformation::{
use crate::anthropic::ANTHROPIC_OAUTH_TOKEN_PREFIX;
use crate::anthropic::experimental_pass_through::messages::transformation::{
complete_anthropic_url, resolve_anthropic_api_key,
};
use crate::llms::base_llm::chat::transformation::{
use crate::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use crate::chat::Error;
use crate::chat::conversation::{Conversation, build_conversation};
use crate::chat::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
ProviderChatRequestData, ProviderChatResponseData,
};
/// Anthropic parameter names, post `map_openai_params`, that the Rust path can
/// place verbatim in the Messages body.

View file

@ -1,4 +1,4 @@
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::messages::Error;
const ANTHROPIC_API_KEY_ENV: &str = "ANTHROPIC_API_KEY";

View file

@ -0,0 +1 @@
pub mod messages;

View file

@ -0,0 +1,4 @@
pub mod chat;
pub mod experimental_pass_through;
pub const ANTHROPIC_OAUTH_TOKEN_PREFIX: &str = "sk-ant-oat";

View file

@ -0,0 +1,31 @@
use thiserror::Error;
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("expected {expected}, got {actual}")]
InvalidType {
expected: &'static str,
actual: &'static str,
},
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub fn json_type_name(value: &serde_json::Value) -> &'static str {
match value {
serde_json::Value::Null => "null",
serde_json::Value::Bool(_) => "boolean",
serde_json::Value::Number(_) => "number",
serde_json::Value::String(_) => "string",
serde_json::Value::Array(_) => "array",
serde_json::Value::Object(_) => "object",
}
}
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::audio_transcription::transformation::{
use crate::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
@ -20,15 +20,15 @@ pub struct AudioTranscriptionRequest<'a> {
#[derive(Clone)]
pub struct ProviderAudioTranscriptionRequest {
pub(super) model: String,
pub(super) custom_llm_provider: String,
pub(super) config: &'static dyn BaseAudioTranscriptionConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) auth: AudioTranscriptionAuth,
pub(super) optional_params: Map<String, Value>,
pub(super) timeout: Option<Duration>,
pub model: String,
pub custom_llm_provider: String,
pub config: &'static dyn BaseAudioTranscriptionConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub auth: AudioTranscriptionAuth,
pub optional_params: Map<String, Value>,
pub timeout: Option<Duration>,
}
impl ProviderAudioTranscriptionRequest {

View file

@ -1,9 +1,9 @@
use serde_json::{Map, Value};
use crate::llms::anthropic::experimental_pass_through::messages::transformation::{
use crate::anthropic::experimental_pass_through::messages::transformation::{
ANTHROPIC_MESSAGES_CONFIG, AnthropicMessagesConfig, non_empty,
};
use crate::llms::base_llm::anthropic_messages::transformation::{
use crate::base_llm::anthropic_messages::transformation::{
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
};
use crate::messages::Error;

View file

@ -0,0 +1 @@
pub mod anthropic;

View file

@ -0,0 +1 @@
pub mod transformation;

View file

@ -0,0 +1 @@
pub mod transformation;

View file

@ -1,7 +1,7 @@
use serde_json::{Map, Value};
use crate::chat_completions::Error;
use crate::chat_completions::types::{
use crate::chat::Error;
use crate::chat::types::{
ChatCompletionsResponse, ChatMessage, ChatMessageContent, ProviderChatRequestData,
ProviderChatResponseData,
};

View file

@ -0,0 +1,3 @@
pub mod anthropic_messages;
pub mod audio_transcription;
pub mod chat;

View file

@ -1,11 +1,11 @@
use serde_json::{Map, Value, json};
use crate::audio_transcription::Error;
use crate::audio_transcription::json_type_name;
use crate::audio_transcription::types::{
AudioTranscriptionRequestData, AudioTranscriptionResponseData,
};
use crate::http_utils::json_type_name;
use crate::llms::base_llm::audio_transcription::transformation::{
use crate::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
use litellm_auth_aws::constants::{BEDROCK_RUNTIME_ENDPOINT_TEMPLATE, BEDROCK_SERVICE};

View file

@ -1,16 +1,16 @@
use serde_json::{Map, Value, json};
use crate::chat_completions::Error;
use crate::chat_completions::conversation::{Conversation, TurnRole, build_conversation};
use crate::chat_completions::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat_completions::types::{
use crate::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use crate::chat::Error;
use crate::chat::conversation::{Conversation, TurnRole, build_conversation};
use crate::chat::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse,
ChatCompletionsUsage, ChatMessage, ChatMessageContent, ProviderChatRequestData,
ProviderChatResponseData,
};
use crate::llms::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use litellm_auth_aws::constants::{AWS_BEARER_TOKEN_BEDROCK, BEDROCK_RUNTIME_ENDPOINT_TEMPLATE};
use litellm_auth_aws::{bedrock_model_id_and_region, resolve_bedrock_region};

View file

@ -1,7 +1,7 @@
use serde_json::json;
use super::*;
use crate::chat_completions::Error;
use crate::chat::Error;
fn messages(value: Value) -> Vec<ChatMessage> {
serde_json::from_value(value).expect("valid messages")

View file

@ -11,7 +11,7 @@
//! accepts; anything richer is declined upstream by the capability gate.
use super::types::{ChatMessage, ChatMessageContent};
use crate::constants::EMPTY_TEXT_PLACEHOLDER;
use crate::chat::EMPTY_TEXT_PLACEHOLDER;
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum TurnRole {

View file

@ -0,0 +1,21 @@
use thiserror::Error;
pub const EMPTY_TEXT_PLACEHOLDER: &str = " ";
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error("unsupported: {0}")]
Unsupported(&'static str),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub mod conversation;
pub mod response_utils;
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
use crate::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
/// A `/chat/completions` call as it crosses into the core.
///
@ -22,26 +22,26 @@ pub struct ChatCompletionsRequest<'a> {
pub timeout: Option<Duration>,
}
pub(super) struct ResolvedChatCompletionsRequest<'a> {
pub(super) model: String,
pub(super) config: &'static dyn BaseConfig,
pub(super) messages: Vec<ChatMessage>,
pub(super) optional_params: Map<String, Value>,
pub(super) api_key: Option<&'a str>,
pub(super) api_base: Option<&'a str>,
pub(super) extra_headers: Option<Map<String, Value>>,
pub(super) timeout: Option<Duration>,
pub struct ResolvedChatCompletionsRequest<'a> {
pub model: String,
pub config: &'static dyn BaseConfig,
pub messages: Vec<ChatMessage>,
pub optional_params: Map<String, Value>,
pub api_key: Option<&'a str>,
pub api_base: Option<&'a str>,
pub extra_headers: Option<Map<String, Value>>,
pub timeout: Option<Duration>,
}
pub(super) struct ProviderChatCompletionsRequest {
pub(super) model: String,
pub(super) config: &'static dyn BaseConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) auth: ChatCompletionsAuth,
pub(super) optional_params: Map<String, Value>,
pub(super) timeout: Option<Duration>,
pub struct ProviderChatCompletionsRequest {
pub model: String,
pub config: &'static dyn BaseConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub auth: ChatCompletionsAuth,
pub optional_params: Map<String, Value>,
pub timeout: Option<Duration>,
}
/// The provider-shaped request body a config produces. Named rather than a bare

View file

@ -0,0 +1,8 @@
pub mod anthropic;
pub mod audio_transcription;
pub mod azure_ai;
pub mod base_llm;
pub mod bedrock;
pub mod chat;
pub mod messages;
pub mod provider_resolution;

View file

@ -0,0 +1,17 @@
use thiserror::Error;
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error("unsupported: {0}")]
Unsupported(&'static str),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
pub struct MessagesRequest<'a> {
pub model: &'a str,
@ -15,14 +15,14 @@ pub struct MessagesRequest<'a> {
pub timeout: Option<Duration>,
}
pub(super) struct ProviderMessagesRequest {
pub(super) provider: String,
pub(super) model: String,
pub(super) config: &'static dyn BaseAnthropicMessagesConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) timeout: Option<Duration>,
pub struct ProviderMessagesRequest {
pub provider: String,
pub model: String,
pub config: &'static dyn BaseAnthropicMessagesConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub timeout: Option<Duration>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]

View file

@ -0,0 +1,33 @@
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct CustomLlmProvider<'a> {
pub model: &'a str,
pub custom_llm_provider: &'a str,
}
pub fn get_custom_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
) -> Option<CustomLlmProvider<'a>> {
if let Some(custom_llm_provider) = custom_llm_provider.filter(|provider| !provider.is_empty()) {
return Some(CustomLlmProvider {
model: strip_custom_llm_provider_prefix(model, custom_llm_provider),
custom_llm_provider,
});
}
let (custom_llm_provider, model) = model.split_once('/')?;
if custom_llm_provider.is_empty() || model.is_empty() {
return None;
}
Some(CustomLlmProvider {
model,
custom_llm_provider,
})
}
fn strip_custom_llm_provider_prefix<'a>(model: &'a str, custom_llm_provider: &str) -> &'a str {
model
.strip_prefix(custom_llm_provider)
.and_then(|model| model.strip_prefix('/'))
.unwrap_or(model)
}

View file

@ -90,6 +90,7 @@ from litellm.litellm_core_utils.logging_utils import (
truncate_base64_in_messages_async,
)
from litellm.litellm_core_utils.model_param_helper import ModelParamHelper
from litellm.litellm_core_utils.ptu_pricing import is_spilled_over_ptu_request
from litellm.litellm_core_utils.redact_messages import (
redact_message_input_output_from_custom_logger,
redact_message_input_output_from_logging,
@ -1746,8 +1747,14 @@ class Logging(LiteLLMLoggingBaseClass):
if transformed_result is not None:
result = transformed_result
result_hidden_params: Final = getattr(result, "_hidden_params", None) or MappingProxyType({})
result_additional_headers: Final = (
result_hidden_params.get("additional_headers")
if isinstance(result_hidden_params, dict)
else getattr(result_hidden_params, "additional_headers", None)
)
if isinstance(result, (BaseModel, HttpxBinaryResponseContent)) and hasattr(result, "_hidden_params"):
hidden_params: Final = getattr(result, "_hidden_params", {})
hidden_params: Final = result_hidden_params
if (
"response_cost" in hidden_params and hidden_params["response_cost"] is not None
): # use cost if already calculated
@ -1762,8 +1769,17 @@ class Logging(LiteLLMLoggingBaseClass):
router_model_id = self.get_router_model_id()
## RESPONSE COST ##
custom_pricing: Final = use_custom_pricing_for_model(
litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None)
spilled_over: Final = is_spilled_over_ptu_request(
model_info=_deployment_model_info(self.litellm_params if hasattr(self, "litellm_params") else None),
response_headers=self.model_call_details.get("response_headers"),
additional_headers=result_additional_headers,
)
custom_pricing: Final = (
False
if spilled_over
else use_custom_pricing_for_model(
litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None)
)
)
prompt = self._prompt_for_cost_calculation()
@ -5257,6 +5273,18 @@ def _get_custom_logger_settings_from_proxy_server(callback_name: str) -> dict:
return {}
def _deployment_model_info(litellm_params: dict | None) -> Mapping[str, object]:
"""The router-stamped deployment model_info from whichever metadata field carries it."""
if litellm_params is None:
return MappingProxyType({})
for metadata_key in ("metadata", "litellm_metadata"):
if not isinstance(metadata := litellm_params.get(metadata_key), Mapping):
continue
if model_info := metadata.get("model_info"):
return model_info
return MappingProxyType({})
def use_custom_pricing_for_model(litellm_params: dict | None) -> bool:
"""
Check if the model uses custom pricing

View file

@ -14,9 +14,11 @@ from typing import Final
from litellm.secret_managers.main import get_secret_bool
from litellm.types.router import ModelInfo
from litellm.types.utils import CustomPricingLiteLLMParams, MirroredPricingParams
from litellm.types.utils import AzureSpillover, CustomPricingLiteLLMParams, MirroredPricingParams
PTU_COST_ATTRIBUTION_ENV_VAR: Final = "LITELLM_ENABLE_PTU_COST_ATTRIBUTION"
AZURE_SPILLOVER_HEADER: Final = "x-ms-is-spilled-over"
AZURE_SPILLOVER_FROM_HEADER: Final = "x-ms-spillover-from-deployment"
def is_ptu_cost_attribution_enabled() -> bool:
@ -235,3 +237,33 @@ def zeroed_ptu_pricing(
),
}
)
def is_spilled_over_ptu_request(
model_info: Mapping[str, object],
response_headers: Mapping[str, object] | None,
additional_headers: Mapping[str, object] | None,
) -> bool:
"""Whether Azure served this request from pay-as-you-go capacity, so the zeroed PTU rates must not apply."""
if ptu_terms(model_info) is None:
return False
if not is_ptu_cost_attribution_enabled():
return False
return azure_spillover(response_headers, additional_headers) is not None
def azure_spillover(
response_headers: Mapping[str, object] | None,
additional_headers: Mapping[str, object] | None,
) -> AzureSpillover | None:
"""The spillover Azure reports in the response headers, else None."""
for headers, prefix in (
(response_headers, ""),
(additional_headers, "llm_provider-"),
):
if headers is None or str(headers.get(f"{prefix}{AZURE_SPILLOVER_HEADER}")).lower() != "true":
continue
return AzureSpillover(
from_deployment=str(v) if (v := headers.get(f"{prefix}{AZURE_SPILLOVER_FROM_HEADER}")) is not None else None
)
return None

View file

@ -561,6 +561,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM):
headers, response = self.make_sync_azure_openai_chat_completion_request(
azure_client=azure_client, data=data, timeout=timeout
)
logging_obj.model_call_details["response_headers"] = headers
streamwrapper: Final = CustomStreamWrapper(
completion_stream=response,
model=model,

View file

@ -34,6 +34,7 @@ if TYPE_CHECKING:
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
def error_response_text(response: httpx.Response) -> str:
@ -878,9 +879,10 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
Whether Converse ``cachePoint`` blocks may be sent to this model.
Bedrock rejects requests carrying cachePoint blocks for models without prompt
caching support ("You invoked an unsupported model or your request did not allow
prompt caching"), so a model whose cost-map entry does not declare
OpenAI-family models only support implicit caching and never accept explicit
``cachePoint`` blocks. Bedrock rejects requests carrying cachePoint blocks for
models without prompt caching support ("You invoked an unsupported model or your
request did not allow prompt caching"), so a model whose cost-map entry does not declare
``supports_prompt_caching`` must not receive them. A model absent from the map
(an application inference profile ARN, a model newer than the map) keeps emitting
so existing caching setups never silently degrade. ``litellm.utils.supports_prompt_caching``
@ -888,6 +890,8 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
if model is None:
return True
if _OPENAI_FAMILY_MODEL_RE.search(model):
return False
entries: Final = tuple(
entry
for candidate in (model, get_bedrock_base_model(model))

View file

@ -23788,7 +23788,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
"supports_vision": true
},
"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf": {
"input_cost_per_token": 1.2e-06,
@ -24114,7 +24114,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
"supports_vision": true
},
"fireworks_ai/qwen3p7-plus": {
"cache_read_input_token_cost": 8e-08,
@ -69184,6 +69184,27 @@
"supports_reasoning": true,
"supports_vision": true
},
"typesafe/jev-1.13.0": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-latest": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-preview": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"wandb/zai-org/GLM-5.3-Flash": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_token": 1.5e-07,

View file

@ -208,6 +208,7 @@ LAZY_FEATURES: Final[tuple[LazyFeature, ...]] = (
"/nvidia_nim/",
"/openai/",
"/openai_passthrough/",
"/typesafe/",
"/vertex-ai/",
"/vertex_ai/",
"/vllm/",

View file

@ -20373,6 +20373,96 @@
]
}
},
"/typesafe/{endpoint}": {
"get": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__get",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
},
"post": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__post",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
}
},
"/vertex_ai/discovery/{endpoint}": {
"delete": {
"description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`",
@ -38680,8 +38770,7 @@
"required": false,
"schema": {
"default": 10,
"maximum": 100,
"minimum": 1,
"minimum": 0,
"title": "Count",
"type": "integer"
}
@ -39385,8 +39474,7 @@
"required": false,
"schema": {
"default": 10,
"maximum": 100,
"minimum": 1,
"minimum": 0,
"title": "Count",
"type": "integer"
}

View file

@ -51,6 +51,7 @@ from litellm.types.router import RouterErrors, UpdateRouterConfig
from litellm.types.router_weights import validate_router_settings_dict
from litellm.types.secret_managers.main import KeyManagementSystem
from litellm.types.utils import (
AzureSpillover,
CallTypes,
CostBreakdown,
EmbeddingResponse,
@ -483,6 +484,7 @@ class LiteLLMRoutes(enum.Enum):
"/eu.assemblyai",
"/vllm",
"/mistral",
"/typesafe",
"/milvus",
"/gigachat",
"/watsonx",
@ -850,6 +852,7 @@ class LiteLLMRoutes(enum.Enum):
"/team/member_add",
"/team/member_delete",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/member_update",
"/team/{team_id}/member/{user_id}/reset_spend",
"/team/permissions_list",
@ -3895,6 +3898,7 @@ class SpendLogsMetadata(TypedDict):
autorouter_savings: ReadOnly[float | None] # stamped by the logging payload; None = not auto-routed
litellm_gateway_injected_cache: ReadOnly[str | None]
router_metadata: ReadOnly[SpendLogsRouterMetadata | None] # None = deployment not flagged internal_router_model
azure_spillover: ReadOnly[AzureSpillover | None] # None = Azure did not report spillover
class SpendLogsPayload(TypedDict):

View file

@ -17,6 +17,7 @@ if TYPE_CHECKING:
AUTO_ROUTER_LICENSE_FEATURE: Final = "auto_router"
LICENSE_ALL_FEATURES: Final = "*"
AUTO_ROUTER_LICENSE_REMEDY: Final = "A LiteLLM license with the 'auto_router' feature lifts the limit."
@ -153,17 +154,21 @@ class LicenseCheck:
return False
return team_count > _max_teams_in_license
def grants_feature(self, feature: str) -> bool:
if self.airgapped_license_data is None:
return False
allowed_features: Final = self.airgapped_license_data.get("allowed_features")
granted: Final = allowed_features if isinstance(allowed_features, list) else (allowed_features,)
return feature in granted or LICENSE_ALL_FEATURES in granted
def auto_router_capability_limit(self) -> int | None:
"""
How many auto-routers may claim each gated classifier or customization capability:
unlimited (None) only when the signed license lists the auto_router
feature, otherwise one per capability. A license verified through the API carries no
feature list, so it does not lift the limit either.
unlimited (None) only when the signed license lists the auto_router feature or the
"*" wildcard that grants every feature, otherwise one per capability. A license verified
through the API carries no feature list, so it does not lift the limit either.
"""
if self.airgapped_license_data is None:
return 1
allowed_features: Final = self.airgapped_license_data.get("allowed_features")
if isinstance(allowed_features, list) and AUTO_ROUTER_LICENSE_FEATURE in allowed_features:
if self.grants_feature(AUTO_ROUTER_LICENSE_FEATURE):
return None
return 1

View file

@ -31,6 +31,7 @@ _PROXY_ADMIN_VIEW_ONLY_BLOCKED_ROUTES: Final = frozenset(
# team
"/team/new",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/update",
"/team/delete",
"/team/block",
@ -767,6 +768,7 @@ class RouteChecks:
"/user/bulk_update",
"/team/new",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/update",
"/team/delete",
"/model/new",

View file

@ -569,15 +569,15 @@ lite --base-url https://your-proxy.example.com configure claude --api-key sk-...
claude
```
The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-<UTF-8 hex of the group name>` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute up` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control
The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-<UTF-8 hex of the group name>` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute start` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control
Plain `lite configure`, with no agent named, asks which agents to wire and which gateway model each starts on, picked from `/v1/models` with a type-to-filter prompt. All choices and selected config files are checked before the first settings write. If a later filesystem write fails, the output identifies each agent already configured and its undo command
What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute up` session holds a backup, and that check comes before any request
What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute start` session holds a backup, and that check comes before any request
#### Routed model and savings in the status line
`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute up` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline:
`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute start` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline:
```
Routed to: claude-haiku-4-5 -63% vs Claude Opus 5
@ -597,7 +597,7 @@ After upgrading the CLI, rerun your original `lite configure claude` command wit
#### Install the CLI
`lite autoroute up` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
`lite autoroute start` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
```bash
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install.sh | sh
@ -610,7 +610,7 @@ curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch-or-commit>/
LITELLM_CLI_REF=<branch-or-commit> sh
```
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute up`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute start`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
Point the CLI at your real proxy and key before running any `lite model-groups` or `lite autoroute` command -- like every other command in this CLI, they read `LITELLM_PROXY_URL`/`LITELLM_PROXY_API_KEY` (or `--base-url`/`--api-key`), no `lite login` required:
@ -637,44 +637,46 @@ An interactive wizard. It runs the same model-group discovery as above, splits t
The wizard writes the result to `~/.litellm/autorouter/config.yaml` with `0600` permissions, since the file embeds your real proxy API key. Every model referenced anywhere in that config -- tier targets, the classifier model, the embedding model -- becomes its own `litellm_proxy/<model-name>` deployment whose `api_base` and `api_key` point back at your real proxy. That is the trick that keeps your real proxy's config untouched: every actual network call this generates, whether it is the routed completion, an LLM-classifier call, or an embedding call, forwards transparently through your real, already-running proxy with your real key.
You do not need to tell Claude Code to request `autorouter` by name yourself: `lite autoroute up` also sets the top-level `model` and `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`, `ANTHROPIC_DEFAULT_OPUS_MODEL` and `ANTHROPIC_DEFAULT_FABLE_MODEL` to `autorouter` in `~/.claude/settings.json` (and `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when missing, like every other wiring), so every one of Claude Code's own model tiers requests it directly regardless of `/model` or whatever it defaults to otherwise. (A bare `model_name: "*"` deployment looks like the obvious way to catch any request instead, but litellm's Router looks up auto-router deployments by the literal requested model string with no wildcard resolution, so a `"*"` entry would never actually match real traffic -- these env var overrides are what makes it work.)
You do not need to tell Claude Code to request `autorouter` by name yourself: `lite autoroute start` also sets the top-level `model` and `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`, `ANTHROPIC_DEFAULT_OPUS_MODEL` and `ANTHROPIC_DEFAULT_FABLE_MODEL` to `autorouter` in `~/.claude/settings.json` (and `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when missing, like every other wiring), so every one of Claude Code's own model tiers requests it directly regardless of `/model` or whatever it defaults to otherwise. (A bare `model_name: "*"` deployment looks like the obvious way to catch any request instead, but litellm's Router looks up auto-router deployments by the literal requested model string with no wildcard resolution, so a `"*"` entry would never actually match real traffic -- these env var overrides are what makes it work.)
You must run `configure` at least once before `up`; running `up` first fails with a clear error telling you to configure first.
You must run `configure` at least once before `start`; running `start` first fails with a clear error telling you to configure first.
#### Launch the Ephemeral Auto-Router Proxy
```bash
lite autoroute up
lite autoroute start
```
Starts a local, throwaway litellm proxy on `127.0.0.1:5483` (override with `--port`), running the config `configure` generated, with a self-issued API key baked in (your real proxy key never leaves the generated config -- it only appears there, forwarding to your real proxy). Both the port and the key are stable across runs: the key is minted once, persisted inside the generated config, and reused by every later `up` (and carried forward when you re-run `configure`), so anything you configured against one session keeps working in the next. If the port is already taken, `up` refuses with a clear error instead of silently moving to another one. It waits for the ephemeral proxy to report healthy, then patches `~/.claude/settings.json` the same way `lite up` does, except with a static `ANTHROPIC_AUTH_TOKEN` env var instead of an `apiKeyHelper`, since this key is self-issued rather than something needing SSO refresh. Any `claude` session started afterward, from any terminal, routes through the ephemeral proxy.
Starts a local, throwaway litellm proxy on `127.0.0.1:5483` (override with `--port`), running the config `configure` generated, with a self-issued API key baked in (your real proxy key never leaves the generated config -- it only appears there, forwarding to your real proxy). Both the port and the key are stable across runs: the key is minted once, persisted inside the generated config, and reused by every later `start` (and carried forward when you re-run `configure`), so anything you configured against one session keeps working in the next. If the port is already taken, `start` refuses with a clear error instead of silently moving to another one. It waits for the ephemeral proxy to report healthy, then patches `~/.claude/settings.json` the same way `lite up` does, except with a static `ANTHROPIC_AUTH_TOKEN` env var instead of an `apiKeyHelper`, since this key is self-issued rather than something needing SSO refresh. Any `claude` session started afterward, from any terminal, routes through the ephemeral proxy.
`lite autoroute up` runs in the foreground and streams the ephemeral proxy's own log file into your terminal, so you can watch its routing decisions -- which tier and model got picked for each request -- as you use Claude Code normally. Press Ctrl-C (or send SIGTERM) to stop it; this kills the child proxy process and restores your original Claude Code settings, in that order.
`lite autoroute start` runs in the foreground and streams the ephemeral proxy's own log file into your terminal, so you can watch its routing decisions -- which tier and model got picked for each request -- as you use Claude Code normally. Press Ctrl-C (or send SIGTERM) to stop it; this kills the child proxy process and restores your original Claude Code settings, in that order.
#### Recover From an Unclean Shutdown
```bash
lite autoroute down
lite autoroute stop
```
If the `lite autoroute up` process dies uncleanly -- `kill -9`, a crash -- rather than being stopped with Ctrl-C, `down` is the manual recovery path: it kills any leftover ephemeral proxy process found via a recorded pid file and restores Claude Code's settings from whatever backup is on disk.
If the `lite autoroute start` process dies uncleanly -- `kill -9`, a crash -- rather than being stopped with Ctrl-C, `stop` is the manual recovery path: it kills any leftover ephemeral proxy process found via a recorded pid file and restores Claude Code's settings from whatever backup is on disk.
#### Example
```bash
lite autoroute configure
lite autoroute up
lite autoroute start
# use Claude Code as normal in another terminal; routing decisions stream live
lite autoroute down # only needed if `up` was killed uncleanly instead of Ctrl-C'd
lite autoroute stop # only needed if `start` was killed uncleanly instead of Ctrl-C'd
```
The previous names, `lite autoroute up` and `lite autoroute down`, still work as hidden aliases of `start` and `stop`: each prints a deprecation notice on stderr and will be removed in a future release
#### Caveats
Adaptive mode's learned state does not persist across `lite autoroute up` sessions -- there is no local database, so every session starts adaptive selection cold. A Claude Code session already running before `up` started, or still running when it stops, keeps whatever settings it loaded at its own startup; like `lite up`, this is a one-time file patch and restore, not a live traffic interceptor. Only Claude Code is supported, for the same reason as `lite up`: no other supported agent (for example Cursor) has an equivalent hot-patchable config file.
Adaptive mode's learned state does not persist across `lite autoroute start` sessions -- there is no local database, so every session starts adaptive selection cold. A Claude Code session already running before `start` ran, or still running when it stops, keeps whatever settings it loaded at its own startup; like `lite up`, this is a one-time file patch and restore, not a live traffic interceptor. Only Claude Code is supported, for the same reason as `lite up`: no other supported agent (for example Cursor) has an equivalent hot-patchable config file.
A session that outlives `up` (or is still running the moment you stop it) keeps sending requests, master key included, to that now-freed loopback port until you restart it. Once the ephemeral proxy process exits, nothing stops another local account on the same machine from binding that same port and receiving those requests instead -- and since the port is a fixed, predictable default and the master key is a static value that persists across sessions (unlike `lite up`'s `apiKeyHelper`, which is re-resolved per request), whoever receives them gets a live-looking token along with the prompt content. Restart any Claude Code session before you consider the machine clean, run `lite autoroute down` promptly rather than leaving a stopped session's settings patched, and do not run `lite autoroute up` on a shared or multi-tenant host. To rotate the persisted key, delete the `master_key` line from `~/.litellm/autorouter/config.yaml`; the next `up` mints a fresh one (deleting the whole file works too, but then `configure` must be re-run first).
A session that outlives `start` (or is still running the moment you stop it) keeps sending requests, master key included, to that now-freed loopback port until you restart it. Once the ephemeral proxy process exits, nothing stops another local account on the same machine from binding that same port and receiving those requests instead -- and since the port is a fixed, predictable default and the master key is a static value that persists across sessions (unlike `lite up`'s `apiKeyHelper`, which is re-resolved per request), whoever receives them gets a live-looking token along with the prompt content. Restart any Claude Code session before you consider the machine clean, run `lite autoroute stop` promptly rather than leaving a stopped session's settings patched, and do not run `lite autoroute start` on a shared or multi-tenant host. To rotate the persisted key, delete the `master_key` line from `~/.litellm/autorouter/config.yaml`; the next `start` mints a fresh one (deleting the whole file works too, but then `configure` must be re-run first).
Do not run `lite up` and `lite autoroute up` at the same time. Each patches `~/.claude/settings.json` and keeps its own separate backup, with no coordination between them: whichever one you stop or crash out of last is the one whose backup gets restored, which can silently leave the *other* mode's settings (a static master key and a now-dead loopback URL, or a stale `apiKeyHelper`) active. Run `lite down` or `lite autoroute down` (whichever applies) before switching to the other mode.
Do not run `lite up` and `lite autoroute start` at the same time. Each patches `~/.claude/settings.json` and keeps its own separate backup, with no coordination between them: whichever one you stop or crash out of last is the one whose backup gets restored, which can silently leave the *other* mode's settings (a static master key and a now-dead loopback URL, or a stale `apiKeyHelper`) active. Run `lite down` or `lite autoroute stop` (whichever applies) before switching to the other mode.
## Environment Variables

View file

@ -1,5 +1,5 @@
"""CLI package for LiteLLM Proxy Client."""
from .main import cli
from .main import cli, litellm_proxy_cli
__all__ = ["cli"]
__all__ = ["cli", "litellm_proxy_cli"]

View file

@ -51,7 +51,7 @@ def _ensure_master_key() -> str:
The generated config is the single home of the key: the proxy server authenticates against
general_settings.master_key only (a key under litellm_settings is silently ignored, which
would leave the ephemeral proxy with no real auth), and the file is written 0600 via
secure_create. Reusing that persisted value keeps the key stable across `up` runs, so a
secure_create. Reusing that persisted value keeps the key stable across `start` runs, so a
client configured against one session keeps working in the next.
"""
with open(CONFIG_PATH, "r") as f:
@ -88,15 +88,18 @@ def configure(ctx: click.Context) -> None:
run_configure_wizard(ctx)
@autoroute_group.command("up")
@click.option(
_PORT_OPTION: Final = click.option(
"--port",
type=click.IntRange(1, 65535),
default=DEFAULT_AUTOROUTE_PORT,
show_default=True,
help="Loopback port for the ephemeral proxy; stable across runs so configured clients keep working.",
)
def up(port: int) -> None:
@autoroute_group.command("start")
@_PORT_OPTION
def start(port: int) -> None:
"""Launch the ephemeral auto-router proxy and route Claude Code through it"""
if not CONFIG_PATH.exists():
raise click.ClickException("No config found. Run `lite autoroute configure` first.")
@ -104,7 +107,7 @@ def up(port: int) -> None:
missing: Final = missing_proxy_runtime_modules()
if missing:
raise click.ClickException(
"lite autoroute up launches a local litellm proxy, which needs the proxy runtime that the "
"lite autoroute start launches a local litellm proxy, which needs the proxy runtime that the "
f"thin `litellm[cli]` install does not include (missing: {', '.join(missing)}). Install the "
"proxy runtime with `uv tool install --force 'litellm[proxy]'`, or to QA a branch, "
"`curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch>/scripts/install.sh | "
@ -117,14 +120,14 @@ def up(port: int) -> None:
raise click.ClickException(str(e))
if existing_pid is not None and is_running(existing_pid.pid):
raise click.ClickException(
"An ephemeral proxy is already running (lite autoroute up looks already active). "
"Run `lite autoroute down` first."
"An ephemeral proxy is already running (lite autoroute start looks already active). "
"Run `lite autoroute stop` first."
)
if AUTOROUTE_BACKUP_PATH.exists():
raise click.ClickException(
f"{AUTOROUTE_BACKUP_PATH} already exists -- `lite autoroute up` looks like it's already "
"running (or crashed without cleanup). Run `lite autoroute down` first."
f"{AUTOROUTE_BACKUP_PATH} already exists -- `lite autoroute start` looks like it's already "
"running (or crashed without cleanup). Run `lite autoroute stop` first."
)
if port == 4000:
@ -135,8 +138,8 @@ def up(port: int) -> None:
if not is_port_available(port):
raise click.ClickException(
f"Port {port} on 127.0.0.1 is already in use. If a previous `lite autoroute up` is still "
"running or crashed, run `lite autoroute down`; otherwise pick a different port with --port."
f"Port {port} on 127.0.0.1 is already in use. If a previous `lite autoroute start` is still "
"running or crashed, run `lite autoroute stop`; otherwise pick a different port with --port."
)
master_key: Final = _ensure_master_key()
@ -196,7 +199,7 @@ def up(port: int) -> None:
click.echo("\nStopped ephemeral proxy and restored Claude Code settings.")
click.echo(
f"Restart any Claude Code session still open from this session, or another local account could "
f"bind the now-free port {port} and receive its requests. Do not use `lite autoroute up` on a "
f"bind the now-free port {port} and receive its requests. Do not use `lite autoroute start` on a "
f"shared or multi-tenant host."
)
@ -214,13 +217,13 @@ def up(port: int) -> None:
_teardown()
@autoroute_group.command("down")
def down() -> None:
@autoroute_group.command("stop")
def stop() -> None:
"""Restore Claude Code settings and stop a leftover ephemeral proxy, if any"""
try:
record: PidRecord | None = read_pid_record()
except ClaudeSettingsError as e:
# down is the crash-recovery path -- a corrupt pid record must not block it; clear the
# stop is the crash-recovery path -- a corrupt pid record must not block it; clear the
# unusable record and keep going rather than leaving the user with no way to clean up.
click.echo(f"{e} Clearing it and continuing cleanup.", err=True)
record = None
@ -238,7 +241,34 @@ def down() -> None:
elif restored.existed:
click.echo(f"Restored {CLAUDE_SETTINGS_PATH} to its original contents.")
else:
click.echo(f"Removed {CLAUDE_SETTINGS_PATH} (it did not exist before `lite autoroute up`).")
click.echo(f"Removed {CLAUDE_SETTINGS_PATH} (it did not exist before `lite autoroute start`).")
AUTOROUTE_ALIAS_DEPRECATION_NOTICE: Final = (
"`lite autoroute {retired}` is deprecated and will be removed in a future release; "
"run `lite autoroute {current}` instead, it takes the same options."
)
def _warn_deprecated_alias(retired: str, current: str) -> None:
click.secho(AUTOROUTE_ALIAS_DEPRECATION_NOTICE.format(retired=retired, current=current), err=True, fg="yellow")
@autoroute_group.command("up", hidden=True)
@_PORT_OPTION
@click.pass_context
def up(ctx: click.Context, port: int) -> None:
"""Deprecated alias of `lite autoroute start`"""
_warn_deprecated_alias("up", "start")
ctx.invoke(start, port=port)
@autoroute_group.command("down", hidden=True)
@click.pass_context
def down(ctx: click.Context) -> None:
"""Deprecated alias of `lite autoroute stop`"""
_warn_deprecated_alias("down", "stop")
ctx.invoke(stop)
__all__ = ["autoroute_group"]

View file

@ -214,7 +214,7 @@ def build_generated_proxy_config(config: AutorouteConfig, master_key: str) -> di
def master_key_from_config(config: dict[str, JsonValue]) -> str | None:
"""The master key persisted in a generated config, or None when absent or blank.
Single definition of "this config already has a usable key", shared by `up` (reuse
Single definition of "this config already has a usable key", shared by `start` (reuse
instead of minting) and the configure wizard (carry the key forward on rewrite) so the
two sites can never disagree on what counts as one. Returned verbatim, never stripped:
the proxy authenticates against the exact bytes under general_settings.master_key, so a

View file

@ -43,12 +43,12 @@ _PROXY_RUNTIME_MODULES: tuple[str, ...] = ("fastapi", "uvicorn", "backoff", "orj
def missing_proxy_runtime_modules() -> tuple[str, ...]:
"""Proxy-server modules that ``lite autoroute up`` needs but the thin CLI install lacks.
"""Proxy-server modules that ``lite autoroute start`` needs but the thin CLI install lacks.
``launch_proxy`` runs the full ``litellm.proxy.proxy_cli`` server, whose dependencies live in
the ``proxy`` extra, not the ``cli`` extra that installs the ``lite`` command. On a thin
``litellm[cli]`` install the subprocess dies with a bare ``ModuleNotFoundError``; detecting the
gap here lets ``up`` fail with an actionable message instead.
gap here lets ``start`` fail with an actionable message instead.
"""
return tuple(name for name in _PROXY_RUNTIME_MODULES if importlib.util.find_spec(name) is None)

View file

@ -94,7 +94,7 @@ def _load_persisted_master_key(config_path: Path) -> str | None:
"""The master key from an existing generated config, so a rewrite carries it forward.
Lenient on a missing or corrupt file: configure is the regeneration path, so it must
succeed from any prior state; a key that cannot be read is simply not carried and `up`
succeed from any prior state; a key that cannot be read is simply not carried and `start`
mints a fresh one.
"""
if not config_path.exists():

View file

@ -1,6 +1,6 @@
"""Shared handling of Claude Code's ~/.claude/settings.json.
`lite up` and `lite autoroute up` patch this file temporarily and restore it on
`lite up` and `lite autoroute start` patch this file temporarily and restore it on
exit; `lite configure claude` patches it persistently and records how to undo it.
All of them need the same merge, and `up` already imports from `auth`, so the
shared parts live here rather than in any one command module. The credential is
@ -88,7 +88,7 @@ class SettingsFileOwner:
SETTINGS_FILE_OWNERS: Final = (
SettingsFileOwner(BACKUP_PATH, "lite up", "lite down"),
SettingsFileOwner(AUTOROUTE_BACKUP_PATH, "lite autoroute up", "lite autoroute down"),
SettingsFileOwner(AUTOROUTE_BACKUP_PATH, "lite autoroute start", "lite autoroute stop"),
)
_SETTINGS_ADAPTER: Final = TypeAdapter(dict[str, JsonValue])
@ -111,7 +111,7 @@ def _is_default_settings_file(settings_path: Path) -> bool:
def settings_file_owners(settings_path: Path) -> tuple[SettingsFileOwner, ...]:
"""The commands whose backups guard settings_path: `lite up` and `lite autoroute up` only ever manage the default file."""
"""The commands whose backups guard settings_path: `lite up` and `lite autoroute start` only ever manage the default file."""
return SETTINGS_FILE_OWNERS if _is_default_settings_file(settings_path) else ()
@ -240,7 +240,7 @@ def _env_object(settings: Mapping[str, JsonValue], path: Path) -> Mapping[str, J
def refuse_while_owned(settings_path: Path, owners: Sequence[SettingsFileOwner]) -> None:
"""Refuse while `lite up` or `lite autoroute up` holds a backup it will restore over any write; a
"""Refuse while `lite up` or `lite autoroute start` holds a backup it will restore over any write; a
purely local check, so commands run it before any login prompt or request."""
for owner in owners:
if owner.backup_path.exists():
@ -262,7 +262,7 @@ def _write_target(settings_path: Path) -> Path:
def write_claude_settings(settings_path: Path, settings: Mapping[str, JsonValue]) -> None:
"""The one way a settings document lands on disk: staged owner-only beside the target and renamed into
place, through a symlink rather than over it. Every writer (`configure`, `up`, `autoroute up` and the
place, through a symlink rather than over it. Every writer (`configure`, `up`, `autoroute start` and the
restores) may be carrying the credential, so none creates the file under the umask or truncates it."""
target: Final = _write_target(settings_path)
try:
@ -341,7 +341,7 @@ def merge_claude_settings(
an apiKeyHelper) are removed, since Claude Code given two credentials may send the wrong one.
ENABLE_TOOL_SEARCH and CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY get their defaults only when
missing. `default_model` is the top-level `model` and env.ANTHROPIC_MODEL (see StartOn);
`tier_model` is `lite autoroute up`'s knob that points every ANTHROPIC_DEFAULT_*_MODEL at one
`tier_model` is `lite autoroute start`'s knob that points every ANTHROPIC_DEFAULT_*_MODEL at one
group. Apart from those tier keys, exactly OWNED_PATHS are touched.
"""
raw_env: Final = settings.get(ENV_KEY, {})

View file

@ -56,7 +56,7 @@ _CLAUDE_CODE_VIEW: Final = MappingProxyType(
_MODEL_OPTION_HELP: Final = (
f"Proxy model to set as {STARTING_MODEL_ROLE}. Must be listed on /v1/models for the key; without it, "
"Claude Code keeps its own default and a pin an earlier configure made is let go of. Nothing pins Claude "
"Code's sub-agent or background tiers; `lite autoroute up` is the mode that does."
"Code's sub-agent or background tiers; `lite autoroute start` is the mode that does."
)

View file

@ -36,8 +36,8 @@ def migrate(ctx: click.Context, check_only: bool, dry_run: bool):
resumable; safe to re-run after an interruption.
Examples:
litellm-proxy encryption migrate --check # attestation scan, no writes
litellm-proxy encryption migrate # perform the migration
lite encryption migrate --check # attestation scan, no writes
lite encryption migrate # perform the migration
"""
client: Final = HTTPClient(ctx.obj["base_url"], ctx.obj["api_key"])

View file

@ -168,5 +168,16 @@ cli.add_command(configure_group)
cli.add_command(unconfigure_group)
LITELLM_PROXY_DEPRECATION_NOTICE: Final = (
"The `litellm-proxy` command is deprecated and will be removed in a future release; "
"run `lite` instead, it takes the same commands and options."
)
def litellm_proxy_cli() -> None:
click.secho(LITELLM_PROXY_DEPRECATION_NOTICE, err=True, fg="yellow")
cli()
if __name__ == "__main__":
cli()

View file

@ -78,3 +78,27 @@ def get_budget_reset_time(budget_duration: str) -> datetime:
`BudgetResetSettings` by injection (creation/update endpoints, startup backfill).
"""
return compute_budget_reset_at(budget_duration, get_budget_reset_settings())
def _is_persistable_budget_duration(budget_duration: str) -> bool:
from litellm.litellm_core_utils.duration_parser import duration_in_seconds
try:
if duration_in_seconds(budget_duration) <= 0:
return False
get_budget_reset_time(budget_duration=budget_duration)
except (ValueError, OverflowError):
return False
return True
def budget_duration_error(budget_duration: str | None) -> str | None:
"""Why `budget_duration` cannot be persisted, or None when it is usable.
A non-positive duration resolves to a reset time of "now", which leaves the row
permanently due: the reset job re-reads it every tick and, once enough of them
exist, they fill each batch and starve every other tenant's reset.
"""
if budget_duration is None or _is_persistable_budget_duration(budget_duration):
return None
return f"Invalid budget_duration '{budget_duration}'. Use a format like '1h', '24h', '7d', or '30d'."

View file

@ -22,7 +22,7 @@ from litellm.types.llms.openai import (
BaseLiteLLMOpenAIResponseObject,
ResponsesAPIResponse,
)
from litellm.types.utils import CallTypesLiteral, LLMResponseTypes, SpecialEnums
from litellm.types.utils import ADDRESSED_RESPONSE_ID_FIELD, CallTypesLiteral, LLMResponseTypes, SpecialEnums
if TYPE_CHECKING:
from litellm.caching.caching import DualCache
@ -32,7 +32,6 @@ if TYPE_CHECKING:
_RESPONSES_API_PROVIDER_PREFIX: Final = "/openai"
_RESPONSES_API_CREATE_ROUTES: Final = frozenset({"/v1/responses", "/responses"})
_ADDRESSED_RESPONSE_ID_KEY: Final = "_litellm_addressed_response_id"
_UNMANAGED_RESPONSE_ID_DETAIL: Final = (
"Forbidden. This response id was not issued by this proxy, so the proxy cannot tell who owns it. "
"To let keys address responses this proxy did not issue, set "
@ -132,7 +131,7 @@ class ResponsesIDSecurity(CustomLogger):
if call_type not in responses_api_call_types:
return None
addressed_id_field: Final = "previous_response_id" if call_type == "aresponses" else "response_id"
retained_id: Final = data.get(_ADDRESSED_RESPONSE_ID_KEY)
retained_id: Final = data.get(ADDRESSED_RESPONSE_ID_FIELD)
addressed_id: Final = (
retained_id if isinstance(retained_id, str) and retained_id else data.get(addressed_id_field)
)
@ -140,7 +139,7 @@ class ResponsesIDSecurity(CustomLogger):
return data
authorized_id: Final = self._authorize_response_id(addressed_id, user_api_key_dict)
data[addressed_id_field] = authorized_id
data[_ADDRESSED_RESPONSE_ID_KEY] = addressed_id
data[ADDRESSED_RESPONSE_ID_FIELD] = addressed_id
return data
def _authorize_response_id(

View file

@ -36,6 +36,10 @@ router: Final = APIRouter()
IMAGE_EDIT_NUMERIC_FORM_FIELDS: Final = numeric_form_fields(get_type_hints(ImageEditRequestParams))
IMAGE_ARRAY_FIELD: Final = "image[]"
MASK_ARRAY_FIELD: Final = "mask[]"
BRACKETED_FILE_FIELDS: Final = frozenset({IMAGE_ARRAY_FIELD, MASK_ARRAY_FIELD})
async def uploadfile_to_bytesio(upload: UploadFile) -> io.BytesIO:
"""
@ -244,9 +248,9 @@ async def image_edit_api(
fastapi_response: Response,
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
image: list[UploadFile] | None = File(None),
image_array: list[UploadFile] | None = File(None, alias="image[]"),
image_array: list[UploadFile] | None = File(None, alias=IMAGE_ARRAY_FIELD),
mask: list[UploadFile] | None = File(None),
mask_array: list[UploadFile] | None = File(None, alias="mask[]"),
mask_array: list[UploadFile] | None = File(None, alias=MASK_ARRAY_FIELD),
model: str | None = None,
):
"""
@ -294,12 +298,14 @@ async def image_edit_api(
#########################################################
# Read request body and convert UploadFiles to BytesIO
#########################################################
data: Final = dict(
coerce_numeric_form_fields(
data: Final = {
key: value
for key, value in coerce_numeric_form_fields(
parsed_body=await _read_request_body(request=request),
numeric_fields=IMAGE_EDIT_NUMERIC_FORM_FIELDS,
)
)
).items()
if key not in BRACKETED_FILE_FIELDS
}
image_files: Final = await batch_to_bytesio(image)
mask_files: Final = await batch_to_bytesio(mask)
if image_files:

View file

@ -1,5 +1,6 @@
import math
from collections.abc import Mapping
from types import MappingProxyType
from typing import TYPE_CHECKING, Any, Final, Optional, Union
from fastapi import HTTPException, status
@ -33,23 +34,11 @@ def validate_budget_duration(budget_duration: str | None, status_code: int = 400
enough of them exist, they fill each batch and starve every other tenant's
reset.
"""
if budget_duration is None:
return
from litellm.proxy.common_utils.timezone_utils import budget_duration_error
from litellm.litellm_core_utils.duration_parser import duration_in_seconds
from litellm.proxy.common_utils.timezone_utils import get_budget_reset_time
try:
if duration_in_seconds(budget_duration) <= 0:
raise ValueError("budget_duration must be positive")
get_budget_reset_time(budget_duration=budget_duration)
except (ValueError, OverflowError):
raise HTTPException(
status_code=status_code,
detail={
"error": f"Invalid budget_duration '{budget_duration}'. Use a format like '1h', '24h', '7d', or '30d'."
},
)
error: Final = budget_duration_error(budget_duration)
if error is not None:
raise HTTPException(status_code=status_code, detail={"error": error})
from litellm._logging import verbose_proxy_logger
@ -490,6 +479,33 @@ _TEAM_MEMBER_BUDGET_LIMIT_FIELDS: Final = (
)
MEMBER_BUDGET_PATCH_FIELDS: Final = MappingProxyType(
{
"max_budget_in_team": "max_budget",
"tpm_limit": "tpm_limit",
"rpm_limit": "rpm_limit",
"budget_duration": "budget_duration",
"allowed_models": "allowed_models",
}
)
def _prisma_value(value: object) -> object:
return list(value) if isinstance(value, tuple) else value
def member_budget_patch(source: BaseModel) -> dict[str, Any]:
"""Map the per-member limit fields a request actually set to their budget-table
columns (merge-patch: a sent value updates, an explicit null clears, an absent
field is left untouched)."""
provided: Final = source.model_dump(exclude_unset=True)
return {
column: _prisma_value(provided[request_field])
for request_field, column in MEMBER_BUDGET_PATCH_FIELDS.items()
if request_field in provided
}
def _is_set_budget_value(value: object) -> bool:
if value is None:
return False
@ -513,6 +529,7 @@ async def _upsert_budget_and_membership(
user_api_key_dict: UserAPIKeyAuth,
budget_patch: dict[str, Any],
team_default_budget_id: str | None = None,
shared_budget_ids: frozenset[str] | None = None,
):
"""
Apply a merge-patch of per-member budget fields to a team membership.
@ -527,6 +544,10 @@ async def _upsert_budget_and_membership(
(from team metadata.team_member_budget_id). When the membership still
points at it, we clone-on-write so editing one member's budget does not
mutate the shared default that every other member points at.
``shared_budget_ids`` extends that protection to any other row more than one
membership points at, which a caller patching several members at once has
already counted; a row listed there is cloned rather than written in place.
"""
if not budget_patch:
return
@ -538,10 +559,8 @@ async def _upsert_budget_and_membership(
get_budget_reset_time(budget_duration=duration) if duration is not None else None
)
is_shared_default: Final = (
existing_budget_id is not None
and team_default_budget_id is not None
and existing_budget_id == team_default_budget_id
is_shared_default: Final = existing_budget_id is not None and (
existing_budget_id == team_default_budget_id or existing_budget_id in (shared_budget_ids or frozenset())
)
async def _disconnect():
@ -563,25 +582,25 @@ async def _upsert_budget_and_membership(
)
return
create_data: Final[dict[str, Any]] = {
source_row: Final = (
await tx.litellm_budgettable.find_unique(where={"budget_id": existing_budget_id}) if is_shared_default else None
)
source: Final[Mapping[str, Any]] = source_row.model_dump() if source_row is not None else MappingProxyType({})
create_data: Final[dict[str, Any]] = { # mutable-ok: Prisma create payloads are dict-shaped
"created_by": user_api_key_dict.user_id or "",
"updated_by": user_api_key_dict.user_id or "",
**MappingProxyType(
{f: source[f] for f in _TEAM_MEMBER_BUDGET_LIMIT_FIELDS if _is_set_budget_value(source.get(f))}
),
**write_data,
}
if is_shared_default:
default_budget_row: Final = await tx.litellm_budgettable.find_unique(where={"budget_id": existing_budget_id})
if default_budget_row is not None:
default_budget_dict: Final = default_budget_row.model_dump()
for field in _TEAM_MEMBER_BUDGET_LIMIT_FIELDS:
value = default_budget_dict.get(field)
if _is_set_budget_value(value):
create_data[field] = value
create_data.update(write_data)
if create_data.get("budget_duration") is not None:
create_data["budget_reset_at"] = get_budget_reset_time(budget_duration=create_data["budget_duration"])
else:
# Restarting an inherited window on an unrelated edit hands the member a free period.
carried: Final = source.get("budget_reset_at") if "budget_duration" not in budget_patch else None
if carried is not None:
create_data["budget_reset_at"] = carried
if create_data.get("budget_reset_at") is None:
create_data.pop("budget_reset_at", None)
if not _has_meaningful_budget_limit(create_data):

View file

@ -1,20 +1,23 @@
"""`POST /management/v1/teams/{team_id}/members/bulk_delete`."""
"""`POST /management/v1/teams/{team_id}/members/bulk_delete` and `.../members/bulk_update`."""
from typing import Annotated, Final
from fastapi import APIRouter, Depends
from fastapi import APIRouter, Depends, Header
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.list_api.common import PROBLEM_TYPE_BASE, ManagementProblem, reject_unknown_query_params
from litellm.proxy.management_endpoints.management_v1.common import MANAGEMENT_V1_PREFIX
from litellm.proxy.management_helpers.bulk_team_member_budgets import bulk_update_team_member_budgets
from litellm.proxy.management_helpers.bulk_user_deletion import bulk_remove_team_members
from litellm.proxy.management_helpers.utils import (
management_endpoint_wrapper, # pyright: ignore[reportUnknownVariableType] # legacy decorator is untyped
)
from litellm.types.proxy.management_endpoints.management_v1 import ProblemDetail
from litellm.types.proxy.management_endpoints.team_endpoints import (
BulkTeamMemberBudgetUpdateRequest,
BulkTeamMemberBudgetUpdateResponse,
BulkTeamMemberDeleteRequest,
BulkTeamMemberDeleteResponse,
)
@ -92,3 +95,88 @@ async def bulk_delete_team_members_action(
detail="Failed to remove team members.",
)
)
@router.post(
"/teams/{team_id}/members/bulk_update",
tags=["team management"], # mutable-ok: FastAPI types `tags` as list[str], not Sequence
dependencies=(Depends(user_api_key_auth), Depends(reject_unknown_query_params)),
response_model=BulkTeamMemberBudgetUpdateResponse,
)
@management_endpoint_wrapper
async def bulk_update_team_member_budgets_action(
team_id: str,
data: BulkTeamMemberBudgetUpdateRequest,
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
litellm_changed_by: Annotated[
str | None,
Header(
description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
),
] = None,
) -> BulkTeamMemberBudgetUpdateResponse:
"""
Set per-member limits for up to 500 members of one team in one call. Same
authorization and member addressing as `/team/member_update`: proxy admins, the team's
admins, and admins of the team's organization, with each member named by exactly one of
`user_id` or `user_email`. Unknown body fields are a 422 and an unknown team is a 404.
Each row is a merge patch of that member's limits: a field left out is untouched, a
field sent as null is cleared, and clearing the last limit drops the member back to the
team default. A budget row shared by several memberships, the team default included, is
copied for the member being patched rather than written in place, so one member's new
cap never lands on anybody else.
`data` holds one result per requested member, in request order, carrying the limits in
force after the write. A row is `success: false` with an `error` when it names nobody on
the team or repeats an earlier row. Roles are not part of this route; `/team/member_update`
still owns them.
Example curl:
```
curl --location 'http://0.0.0.0:4000/management/v1/teams/team-1/members/bulk_update' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{"members": [{"user_id": "user-1", "max_budget_in_team": 10}, {"user_email": "user-2@example.com", "max_budget_in_team": 10, "budget_duration": "30d"}]}'
```
"""
try:
from litellm.proxy.proxy_server import litellm_proxy_admin_name, prisma_client, user_api_key_cache
if prisma_client is None:
raise ManagementProblem(
ProblemDetail(
type=f"{PROBLEM_TYPE_BASE}database-not-connected",
title="Database not connected",
status=503,
detail=CommonProxyErrors.db_not_connected_error.value,
)
)
results: Final = await bulk_update_team_member_budgets(
team_id=team_id,
data=data,
user_api_key_dict=user_api_key_dict,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
litellm_proxy_admin_name=litellm_proxy_admin_name,
litellm_changed_by=litellm_changed_by,
)
return BulkTeamMemberBudgetUpdateResponse(data=results)
except ManagementProblem:
raise
except Exception as e: # noqa: BLE001 # a driver error answers as a problem document, not the OpenAI error shape
verbose_proxy_logger.exception(
"litellm.proxy.management_endpoints.management_v1.teams.bulk_update_team_member_budgets_action(): "
"Exception occured - %s",
e,
)
raise ManagementProblem(
ProblemDetail(
type=f"{PROBLEM_TYPE_BASE}internal-server-error",
title="Internal server error",
status=500,
detail="Failed to update team member budgets.",
)
)

View file

@ -264,6 +264,8 @@ scim_router: Final = APIRouter(
dependencies=[Depends(_premium_user_check)],
)
SCIM_MAX_PAGE_SIZE: Final = 100
# Helper functions for common operations
async def _get_prisma_client_or_raise_exception():
@ -1572,12 +1574,13 @@ def _parse_scim_eq_filter(scim_filter: str) -> tuple[str, str] | None:
)
async def get_users(
startIndex: int = Query(1, ge=1),
count: int = Query(10, ge=1, le=100),
count: int = Query(10, ge=0),
filter: str | None = Query(None),
):
"""
Get a list of users according to SCIM v2 protocol
"""
page_size: Final = min(count, SCIM_MAX_PAGE_SIZE)
verbose_proxy_logger.debug(
"SCIM GET USERS request: startIndex=%s count=%s filter=%s",
startIndex,
@ -1607,7 +1610,7 @@ async def get_users(
users: Final[Sequence[LiteLLM_UserTable]] = await _table(UserRepository(prisma_client)).find_many(
where=where_conditions,
skip=(startIndex - 1),
take=count,
take=page_size,
order={"created_at": "desc"},
)
@ -1623,7 +1626,7 @@ async def get_users(
return SCIMListResponse(
totalResults=total_count,
startIndex=startIndex,
itemsPerPage=min(count, len(scim_users)),
itemsPerPage=len(scim_users),
Resources=scim_users,
)
@ -2399,12 +2402,13 @@ class _TeamWhereConditions(TypedDict, total=False):
)
async def get_groups(
startIndex: int = Query(1, ge=1),
count: int = Query(10, ge=1, le=100),
count: int = Query(10, ge=0),
filter: str | None = Query(None),
):
"""
Get a list of groups according to SCIM v2 protocol
"""
page_size: Final = min(count, SCIM_MAX_PAGE_SIZE)
verbose_proxy_logger.debug(
"SCIM GET GROUPS request: startIndex=%s count=%s filter=%s",
startIndex,
@ -2425,7 +2429,7 @@ async def get_groups(
teams: Final = await _table(TeamRepository(prisma_client)).find_many(
where=where_conditions,
skip=(startIndex - 1),
take=count,
take=page_size,
order={"created_at": "desc"},
)
@ -2462,7 +2466,7 @@ async def get_groups(
return SCIMListResponse(
totalResults=total_count,
startIndex=startIndex,
itemsPerPage=min(count, len(scim_groups)),
itemsPerPage=len(scim_groups),
Resources=scim_groups,
)

View file

@ -129,6 +129,7 @@ from litellm.proxy.management_endpoints.common_utils import (
_update_metadata_fields,
_upsert_budget_and_membership,
_user_has_admin_view,
member_budget_patch,
validate_budget_duration,
validate_team_model_max_budget,
)
@ -3686,27 +3687,6 @@ async def team_member_delete(
return existing_team_row
_MEMBER_BUDGET_PATCH_FIELDS: Final = {
"max_budget_in_team": "max_budget",
"tpm_limit": "tpm_limit",
"rpm_limit": "rpm_limit",
"budget_duration": "budget_duration",
"allowed_models": "allowed_models",
}
def _build_member_budget_patch(data: TeamMemberUpdateRequest) -> dict[str, object]:
"""Map the budget fields the request actually set (merge-patch: a sent
value updates, an explicit null clears, an absent field is left untouched)
to their budget-table columns."""
provided: Final = data.model_dump(exclude_unset=True)
return {
column: provided[request_field]
for request_field, column in _MEMBER_BUDGET_PATCH_FIELDS.items()
if request_field in provided
}
@router.post(
"/team/member_update",
tags=["team management"],
@ -3812,7 +3792,7 @@ async def team_member_update(
team_default_budget_id = raw_default_budget_id
### upsert new budget
budget_patch: Final = _build_member_budget_patch(data)
budget_patch: Final = member_budget_patch(data)
async with prisma_client.tx() as tx:
await _upsert_budget_and_membership(
tx=tx,

View file

@ -354,7 +354,7 @@ def _get_cli_sso_flow_or_raise(login_id: str | None, cache: DualCache) -> dict:
status_code=400,
detail=(
"Your litellm CLI is out of date and uses a login flow this proxy no longer supports. "
"Upgrade it with `pip install -U 'litellm[proxy]'` and run `litellm-proxy login` again."
"Upgrade it with `pip install -U 'litellm[proxy]'` and run `lite login` again."
),
)
if not _is_valid_cli_sso_login_id(login_id):
@ -375,7 +375,7 @@ def _get_cli_sso_flow_or_raise(login_id: str | None, cache: DualCache) -> dict:
raise HTTPException(
status_code=400,
detail=(
"CLI login session not found or expired. Run `litellm-proxy login` again. "
"CLI login session not found or expired. Run `lite login` again. "
"If this happens immediately after starting a login, the proxy is likely running multiple "
"replicas without a shared cache; configure a Redis cache "
"so every replica can see the login session."

View file

@ -0,0 +1,263 @@
"""Batched per-member limit writes behind `POST /management/v1/teams/{team_id}/members/bulk_update`.
Every read runs on the writer inside the batch transaction, so the write plan can never be
built from a lagging read replica. Any budget row that more than one membership points at,
the team's shared default included, is cloned before it is written, so raising one member's
cap never moves another member's.
"""
from collections.abc import Sequence
from datetime import datetime, timedelta
from types import MappingProxyType
from typing import TYPE_CHECKING, Final
from pydantic import BaseModel, ConfigDict
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.proxy._types import (
LiteLLM_TeamTable,
LitellmTableNames,
LitellmUserRoles,
Member,
UserAPIKeyAuth,
)
from litellm.proxy.auth.auth_checks import invalidate_team_member_spend_state
from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
from litellm.proxy.db.routing_prisma_wrapper import WriterPinnedClient
from litellm.proxy.management_endpoints.common_utils import (
_is_user_org_admin_for_team, # pyright: ignore[reportPrivateUsage] # same check /team/member_update uses
_is_user_team_admin, # pyright: ignore[reportPrivateUsage] # same check /team/member_update uses
_upsert_budget_and_membership, # pyright: ignore[reportPrivateUsage] # the single-member write, shared so the two surfaces cannot drift
member_budget_patch,
)
from litellm.proxy.management_helpers.audit_logs import create_object_audit_log
from litellm.proxy.management_helpers.bulk_user_deletion import (
_duplicate_member_indexes, # pyright: ignore[reportPrivateUsage] # same duplicate rule as members/bulk_delete
_eq_filter, # pyright: ignore[reportPrivateUsage] # same prisma filter shape as members/bulk_delete
_forbidden, # pyright: ignore[reportPrivateUsage] # same problem shape as members/bulk_delete
_in_filter, # pyright: ignore[reportPrivateUsage] # same prisma filter shape as members/bulk_delete
_team_not_found, # pyright: ignore[reportPrivateUsage] # same problem shape as members/bulk_delete
_team_users_filter, # pyright: ignore[reportPrivateUsage] # same prisma filter shape as members/bulk_delete
)
from litellm.proxy.utils import PrismaClient
from litellm.repositories.team_repository import TeamRepository
from litellm.types.proxy.management_endpoints.team_endpoints import (
BulkTeamMemberBudgetUpdateRequest,
TeamMemberBudgetPatch,
TeamMemberBudgetUpdateResult,
)
if TYPE_CHECKING:
from prisma import Prisma
from prisma import models as prisma_models
from litellm.repositories.prisma_protocols import TableActions
_BATCH_TX_TIMEOUT: Final = timedelta(seconds=60)
_NO_METADATA: Final = MappingProxyType({})
_WITH_BUDGET: Final = MappingProxyType({"litellm_budget_table": True})
def _membership_tx_db(tx: "Prisma") -> "TableActions[prisma_models.LiteLLM_TeamMembership]":
return tx.litellm_teammembership # pyright: ignore[reportReturnType] # TableActions widens the generated inputs to Mapping, as the repositories do
def _budget_tx_db(tx: "Prisma") -> "TableActions[prisma_models.LiteLLM_BudgetTable]":
return tx.litellm_budgettable # pyright: ignore[reportReturnType] # TableActions widens the generated inputs to Mapping, as the repositories do
def _roster_user_id(member: TeamMemberBudgetPatch, roster: Sequence[Member]) -> str | None:
"""The team member this row addresses, or None when it names nobody on the team."""
if member.user_id is not None:
return member.user_id if any(m.user_id == member.user_id for m in roster) else None
return next((m.user_id for m in roster if m.user_email is not None and m.user_email == member.user_email), None)
def _team_default_budget_id(team: LiteLLM_TeamTable) -> str | None:
raw: Final = (team.metadata or _NO_METADATA).get("team_member_budget_id")
return raw if isinstance(raw, str) else None
async def _shared_budget_ids(tx: "Prisma", budget_ids: frozenset[str]) -> frozenset[str]:
"""The rows in ``budget_ids`` more than one membership points at, counted across every
team so a row shared with another team is protected too."""
if not budget_ids:
return frozenset()
rows: Final = await _membership_tx_db(tx).find_many(where=_in_filter("budget_id", budget_ids))
return frozenset(budget_id for budget_id in budget_ids if sum(1 for row in rows if row.budget_id == budget_id) > 1)
class _AuditedMemberBudget(BaseModel):
"""One member's limits as the audit log's before/after values record them."""
model_config = ConfigDict(frozen=True)
user_id: str
budget_id: str | None = None
max_budget: float | None = None
tpm_limit: int | None = None
rpm_limit: int | None = None
budget_duration: str | None = None
budget_reset_at: datetime | None = None
allowed_models: tuple[str, ...] | None = None
class _AuditedMemberBudgets(BaseModel):
"""The audit-log columns hold a JSON object, so the per-member list is nested under a key."""
model_config = ConfigDict(frozen=True)
team_member_budgets: tuple[_AuditedMemberBudget, ...]
def _audited_member_budget(row: "prisma_models.LiteLLM_TeamMembership") -> _AuditedMemberBudget:
budget: Final = row.litellm_budget_table
if budget is None:
return _AuditedMemberBudget(user_id=row.user_id, budget_id=row.budget_id)
return _AuditedMemberBudget(
user_id=row.user_id,
budget_id=row.budget_id,
max_budget=budget.max_budget,
tpm_limit=budget.tpm_limit,
rpm_limit=budget.rpm_limit,
budget_duration=budget.budget_duration,
budget_reset_at=budget.budget_reset_at,
allowed_models=tuple(budget.allowed_models),
)
def _limits_audit_value(rows: "Sequence[prisma_models.LiteLLM_TeamMembership]") -> str:
"""Serialize the members' limits for an audit-log value, dropping the limits they do not set."""
return safe_dumps(
_AuditedMemberBudgets(
team_member_budgets=tuple(_audited_member_budget(row) for row in sorted(rows, key=lambda row: row.user_id))
).model_dump(exclude_none=True, mode="json")
)
def _result(
member: TeamMemberBudgetPatch,
user_id: str | None,
error: str | None,
budget_of: "MappingProxyType[str, prisma_models.LiteLLM_BudgetTable | None]",
team_default_max_budget: float | None,
) -> TeamMemberBudgetUpdateResult:
if error is not None or user_id is None:
return TeamMemberBudgetUpdateResult(
user_id=member.user_id,
user_email=member.user_email,
success=False,
error=error or "User not found in team",
)
budget: Final = budget_of.get(user_id)
own_max_budget: Final = budget.max_budget if budget is not None else None
inherits: Final = own_max_budget is None and team_default_max_budget is not None and team_default_max_budget > 0
return TeamMemberBudgetUpdateResult(
user_id=user_id,
user_email=member.user_email,
success=True,
budget_id=budget.budget_id if budget is not None else None,
max_budget=team_default_max_budget if inherits else own_max_budget,
max_budget_source=("team_default" if inherits else "member" if own_max_budget is not None else None),
tpm_limit=budget.tpm_limit if budget is not None else None,
rpm_limit=budget.rpm_limit if budget is not None else None,
budget_duration=budget.budget_duration if budget is not None else None,
allowed_models=tuple(budget.allowed_models) if budget is not None else None,
)
async def bulk_update_team_member_budgets(
team_id: str,
data: BulkTeamMemberBudgetUpdateRequest,
user_api_key_dict: UserAPIKeyAuth,
prisma_client: PrismaClient,
user_api_key_cache: UserApiKeyCache,
litellm_proxy_admin_name: str,
litellm_changed_by: str | None = None,
) -> tuple[TeamMemberBudgetUpdateResult, ...]:
"""Apply one merge patch of per-member limits per requested member, in one transaction."""
team: Final = await TeamRepository(WriterPinnedClient(prisma_client.db)).find_by_id(team_id)
if team is None:
raise _team_not_found(team_id)
if (
user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value
and not _is_user_team_admin(user_api_key_dict=user_api_key_dict, team_obj=team)
and not await _is_user_org_admin_for_team(user_api_key_dict=user_api_key_dict, team_obj=team)
):
raise _forbidden(
"Call not allowed. User not proxy admin OR team admin OR org admin for this team. "
f"route='/management/v1/teams/{team_id}/members/bulk_update'"
)
roster: Final = team.members_with_roles or ()
named: Final = tuple(_roster_user_id(member, roster) for member in data.members)
duplicates: Final = _duplicate_member_indexes(data.members) | frozenset(
index for index, user_id in enumerate(named) if user_id is not None and user_id in named[:index]
)
applied: Final = tuple(
(index, user_id) for index, user_id in enumerate(named) if user_id is not None and index not in duplicates
)
if not applied:
return tuple(
_result(
member, None, "Duplicate member in request" if index in duplicates else None, MappingProxyType({}), None
)
for index, member in enumerate(data.members)
)
user_ids: Final = sorted(user_id for _, user_id in applied)
default_budget_id: Final = _team_default_budget_id(team)
team_members_filter: Final = _team_users_filter(team_id, user_ids)
async with prisma_client.tx(timeout=_BATCH_TX_TIMEOUT) as tx:
memberships: Final = await _membership_tx_db(tx).find_many(where=team_members_filter, include=_WITH_BUDGET)
budget_id_of: Final = MappingProxyType({m.user_id: m.budget_id for m in memberships})
shared: Final = await _shared_budget_ids(
tx, frozenset(budget_id for budget_id in budget_id_of.values() if budget_id is not None)
)
for index, user_id in applied:
await _upsert_budget_and_membership(
tx=tx,
team_id=team_id,
user_id=user_id,
existing_budget_id=budget_id_of.get(user_id),
user_api_key_dict=user_api_key_dict,
budget_patch=member_budget_patch(data.members[index]),
team_default_budget_id=default_budget_id,
shared_budget_ids=shared,
)
written: Final = await _membership_tx_db(tx).find_many(where=team_members_filter, include=_WITH_BUDGET)
team_default: Final = (
await _budget_tx_db(tx).find_unique(where=_eq_filter("budget_id", default_budget_id))
if default_budget_id is not None
else None
)
for user_id in user_ids:
await invalidate_team_member_spend_state(
user_id=user_id, team_id=team_id, user_api_key_cache=user_api_key_cache
)
await create_object_audit_log(
object_id=team_id,
action="updated",
litellm_changed_by=litellm_changed_by,
user_api_key_dict=user_api_key_dict,
litellm_proxy_admin_name=litellm_proxy_admin_name,
table_name=LitellmTableNames.TEAM_TABLE_NAME,
before_value=_limits_audit_value(memberships),
after_value=_limits_audit_value(written),
)
budget_of: Final = MappingProxyType({m.user_id: m.litellm_budget_table for m in written})
return tuple(
_result(
member,
named[index],
"Duplicate member in request" if index in duplicates else None,
budget_of,
team_default.max_budget if team_default is not None else None,
)
for index, member in enumerate(data.members)
)

View file

@ -525,6 +525,42 @@ async def mistral_proxy_route(
return received_value
@router.api_route(
"/typesafe/{endpoint:path}",
methods=["GET", "POST"], # mutable-ok: FastAPI route metadata requires a list
tags=["TypeSafe AI Pass-through", "pass-through"], # mutable-ok: FastAPI route metadata requires a list
)
async def typesafe_proxy_route(
endpoint: str,
request: Request,
fastapi_response: Response,
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
):
"""[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)"""
base_target_url: Final = get_secret_str("TYPESAFE_API_BASE") or "https://api.typesafe.ai"
encoded_endpoint: Final = httpx.URL(endpoint).path
normalized_endpoint: Final = encoded_endpoint if encoded_endpoint.startswith("/") else f"/{encoded_endpoint}"
base_url: Final = httpx.URL(base_target_url)
updated_url: Final = base_url.copy_with(
path=HttpPassThroughEndpointHelpers.join_base_and_endpoint_path(base_url, normalized_endpoint),
)
typesafe_api_key: Final = passthrough_endpoint_router.get_credentials(
custom_llm_provider="typesafe",
region_name=None,
)
endpoint_func: Final = create_pass_through_route(
endpoint=endpoint,
target=str(updated_url),
custom_headers={ # mutable-ok: pass-through request headers require a mutable mapping
"Authorization": f"Bearer {typesafe_api_key}",
"Content-Type": "application/json",
},
custom_llm_provider="typesafe",
is_streaming_request=False,
)
return await endpoint_func(request, fastapi_response, user_api_key_dict)
@router.api_route(
"/milvus/{endpoint:path}",
methods=["GET", "POST", "PUT", "DELETE", "PATCH"],

View file

@ -0,0 +1,117 @@
from collections.abc import Mapping
from datetime import datetime
from typing import Final
import httpx
from pydantic import BaseModel, TypeAdapter, ValidationError
import litellm
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.litellm_core_utils.litellm_logging import (
get_standard_logging_object_payload, # pyright: ignore[reportUnknownVariableType] # legacy helper has an untyped signature
)
from litellm.proxy._types import PassThroughEndpointLoggingTypedDict
from litellm.types.utils import ModelResponse, StandardPassThroughResponseObject, Usage
class _TypeSafeUsage(BaseModel):
input_tokens: int = 0
output_tokens: int = 0
class _TypeSafeResponse(BaseModel):
model: str | None = None
usage: _TypeSafeUsage | None = None
class _RegistryPricing(BaseModel):
input_cost_per_token: float = 0.0
output_cost_per_token: float = 0.0
_TYPESAFE_RESPONSE_ADAPTER: Final = TypeAdapter(_TypeSafeResponse)
_REGISTRY_PRICING_ADAPTER: Final = TypeAdapter(_RegistryPricing)
def _parse_typesafe_response(response_body: Mapping[str, object]) -> _TypeSafeResponse:
try:
return _TYPESAFE_RESPONSE_ADAPTER.validate_python(response_body)
except ValidationError:
return _TypeSafeResponse()
def _pricing_for(model_keys: tuple[str, ...]) -> _RegistryPricing:
for model_key in model_keys:
if model_key not in litellm.model_cost: # pyright: ignore[reportUnknownMemberType] # registry is dynamically typed
continue
try:
return _REGISTRY_PRICING_ADAPTER.validate_python(
litellm.model_cost[model_key] # pyright: ignore[reportUnknownMemberType] # registry is dynamically typed
)
except ValidationError:
continue
return _RegistryPricing()
class TypeSafePassthroughLoggingHandler:
@staticmethod
def typesafe_passthrough_handler(
httpx_response: httpx.Response,
response_body: Mapping[str, object],
logging_obj: LiteLLMLoggingObj,
url_route: str,
result: str,
start_time: datetime,
end_time: datetime,
cache_hit: bool,
request_body: Mapping[str, object],
**kwargs: object,
) -> PassThroughEndpointLoggingTypedDict:
response: Final = _parse_typesafe_response(response_body)
response_model: Final = response.model
request_model_value: Final = request_body.get("model")
request_model: Final = request_model_value if isinstance(request_model_value, str) else None
logged_model: Final = response_model or request_model or "unknown"
model_name: Final = f"typesafe/{logged_model}"
usage: Final = response.usage or _TypeSafeUsage()
input_tokens: Final = usage.input_tokens
output_tokens: Final = usage.output_tokens
candidate_model_keys: Final = tuple(
f"typesafe/{model}" for model in (response_model, request_model) if model is not None
)
pricing: Final = _pricing_for(candidate_model_keys)
response_cost: Final = (
input_tokens * pricing.input_cost_per_token + output_tokens * pricing.output_cost_per_token
)
usage_object: Final = Usage(
prompt_tokens=input_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + output_tokens,
)
updated_kwargs: Final = { # mutable-ok: pass-through logging contract requires mutable kwargs
**kwargs,
"model": model_name,
"custom_llm_provider": "typesafe",
"response_cost": response_cost,
"combined_usage_object": usage_object,
}
logging_obj.model_call_details.update(
model=model_name,
custom_llm_provider="typesafe",
response_cost=response_cost,
)
standard_logging_object: Final = get_standard_logging_object_payload(
kwargs=updated_kwargs,
init_response_obj=ModelResponse(model=model_name, usage=usage_object),
start_time=start_time,
end_time=end_time,
logging_obj=logging_obj,
status="success",
)
return { # mutable-ok: pass-through logging contract requires mutable result
"result": StandardPassThroughResponseObject(response=result),
"kwargs": { # mutable-ok: pass-through logging contract requires mutable kwargs
**updated_kwargs,
"standard_logging_object": standard_logging_object,
},
}

View file

@ -1,5 +1,6 @@
import json
from datetime import datetime
from types import MappingProxyType
from typing import Any, Final
from urllib.parse import urlparse
@ -256,6 +257,25 @@ class PassThroughEndpointLogging:
)
standard_logging_response_object = comprehend_medical_handler_result["result"] # rebind-ok: elif-chain
kwargs = comprehend_medical_handler_result["kwargs"] # rebind-ok: elif-chain contract
elif self.is_typesafe_route(custom_llm_provider):
from .llm_provider_handlers.typesafe_passthrough_logging_handler import (
TypeSafePassthroughLoggingHandler,
)
typesafe_handler_result: Final = TypeSafePassthroughLoggingHandler.typesafe_passthrough_handler(
httpx_response=httpx_response,
response_body=response_body if isinstance(response_body, dict) else MappingProxyType({}),
logging_obj=logging_obj,
url_route=url_route,
result=result,
start_time=start_time,
end_time=end_time,
cache_hit=cache_hit,
request_body=request_body,
**kwargs,
)
standard_logging_response_object = typesafe_handler_result["result"]
kwargs = typesafe_handler_result["kwargs"]
elif self.is_vertex_ai_live_route(url_route):
from .llm_provider_handlers.vertex_ai_live_passthrough_logging_handler import (
VertexAILivePassthroughLoggingHandler,
@ -389,6 +409,9 @@ class PassThroughEndpointLogging:
def is_comprehend_medical_route(self, custom_llm_provider: str | None) -> bool:
return custom_llm_provider == "comprehendmedical"
def is_typesafe_route(self, custom_llm_provider: str | None) -> bool:
return custom_llm_provider == "typesafe"
def is_langfuse_route(self, url_route: str):
parsed_url: Final = urlparse(url_route)
for route in self.TRACKED_LANGFUSE_ROUTES:

View file

@ -39,6 +39,7 @@ from litellm.litellm_core_utils.litellm_logging import (
is_valid_sha256_hash,
request_model_access_groups_from_litellm_params,
)
from litellm.litellm_core_utils.ptu_pricing import azure_spillover
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps, strip_null_bytes
from litellm.proxy._types import SpendLogsMetadata, SpendLogsPayload, SpendLogsRouterMetadata
from litellm.proxy.route_llm_request import ProxyModelNotFoundError
@ -47,6 +48,7 @@ from litellm.proxy.utils import PrismaClient, hash_token
from litellm.types.router import DeploymentTypedDict, LiteLLM_Params
from litellm.types.utils import (
PROMPT_CARRYING_GUARDRAIL_FIELDS,
AzureSpillover,
CallTypes,
CostBreakdown,
LlmProviders,
@ -133,6 +135,9 @@ def _get_router_metadata_for_spend_log(
)
_STAMPED_METADATA_KEYS: Final = frozenset(("router_metadata", "azure_spillover"))
def _get_spend_logs_metadata(
metadata: dict | None,
applied_guardrails: list[str] | None = None,
@ -150,6 +155,7 @@ def _get_spend_logs_metadata(
litellm_call_id: str | None = None,
autorouter_savings: float | None = None,
router_metadata: SpendLogsRouterMetadata | None = None,
azure_spillover: AzureSpillover | None = None,
) -> SpendLogsMetadata:
if metadata is None:
return SpendLogsMetadata(
@ -191,6 +197,7 @@ def _get_spend_logs_metadata(
litellm_gateway_injected_cache=None,
litellm_call_id=litellm_call_id,
router_metadata=router_metadata,
azure_spillover=azure_spillover,
)
verbose_proxy_logger.debug(
"getting payload for SpendLogs, available keys in metadata: " + str(list(metadata.keys()))
@ -198,8 +205,9 @@ def _get_spend_logs_metadata(
# Filter the metadata dictionary to include only the specified keys
clean_metadata: Final = SpendLogsMetadata(
**{key: metadata.get(key) for key in SpendLogsMetadata.__annotations__ if key != "router_metadata"},
**{key: metadata.get(key) for key in SpendLogsMetadata.__annotations__ if key not in _STAMPED_METADATA_KEYS},
router_metadata=router_metadata,
azure_spillover=azure_spillover,
)
_raw_key: Final = clean_metadata.get("user_api_key")
_trusted_hash: Final = metadata.get("user_api_key_hash")
@ -573,6 +581,15 @@ def get_logging_payload(
selected_provider=custom_llm_provider,
router_correlation_id=litellm_call_id,
),
azure_spillover=azure_spillover(
response_headers=kwargs.get("response_headers")
if isinstance(kwargs.get("response_headers"), Mapping)
else None,
additional_headers=standard_logging_payload["hidden_params"].get("additional_headers")
if standard_logging_payload is not None
and isinstance(standard_logging_payload.get("hidden_params"), Mapping)
else None,
),
)
special_usage_fields: Final = ["completion_tokens", "prompt_tokens", "total_tokens"]

Some files were not shown because too many files have changed in this diff Show more