Merge remote-tracking branch 'origin/main' into litellm_remove_brittle_price_pinning_tests

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

# Conflicts:
#	tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py
This commit is contained in:
kerry 2026-09-18 00:29:52 +00:00
commit 84d4d17bf8
168 changed files with 11076 additions and 1220 deletions

View file

@ -3,7 +3,7 @@
Example: Using CLI token with LiteLLM SDK
This example shows how to use the CLI authentication token
in your Python scripts after running `litellm-proxy login`.
in your Python scripts after running `lite login`.
"""
from textwrap import indent
@ -22,7 +22,7 @@ def main():
api_key = litellm.get_litellm_gateway_api_key()
if not api_key:
print("❌ No CLI token found. Please run 'litellm-proxy login' first.")
print("❌ No CLI token found. Please run 'lite login' first.")
return
print("✅ Found CLI token.")
@ -58,6 +58,6 @@ if __name__ == "__main__":
main()
print("\n💡 Tips:")
print("1. Run 'litellm-proxy login' to authenticate first")
print("1. Run 'lite login' to authenticate first")
print("2. Replace 'https://your-proxy.com' with your actual proxy URL")
print("3. The token is stored in your OS keychain, or in ~/.litellm/token.json when there is none")

View file

@ -1,614 +0,0 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "grafana",
"uid": "-- Grafana --"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"description": "",
"editable": true,
"fiscalYearStartMonth": 0,
"graphTooltip": 0,
"id": 2039,
"links": [],
"liveNow": false,
"panels": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "s"
},
"overrides": []
},
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 0
},
"id": 10,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "histogram_quantile(0.99, sum(rate(litellm_self_latency_bucket{self=\"self\"}[1m])) by (le))",
"legendFormat": "Time to first token",
"range": true,
"refId": "A"
}
],
"title": "Time to first token (latency)",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "currencyUSD"
},
"overrides": [
{
"matcher": {
"id": "byName",
"options": "7e4b0627fd32efdd2313c846325575808aadcf2839f0fde90723aab9ab73c78f"
},
"properties": [
{
"id": "displayName",
"value": "Translata"
}
]
}
]
},
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 8
},
"id": 11,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_spend_metric_total[30d])) by (hashed_api_key)",
"legendFormat": "{{team}}",
"range": true,
"refId": "A"
}
],
"title": "Spend by team",
"transformations": [],
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 9,
"w": 12,
"x": 0,
"y": 16
},
"id": 2,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum by (model) (increase(litellm_requests_metric_total[5m]))",
"legendFormat": "{{model}}",
"range": true,
"refId": "A"
}
],
"title": "Requests by model",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [],
"noValue": "0",
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 3,
"x": 0,
"y": 25
},
"id": 8,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"textMode": "auto"
},
"pluginVersion": "9.4.17",
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_llm_api_failed_requests_metric_total[1h]))",
"legendFormat": "__auto",
"range": true,
"refId": "A"
}
],
"title": "Faild Requests",
"type": "stat"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "currencyUSD"
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 3,
"x": 3,
"y": 25
},
"id": 6,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_spend_metric_total[30d])) by (model)",
"legendFormat": "{{model}}",
"range": true,
"refId": "A"
}
],
"title": "Spend",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 6,
"x": 6,
"y": 25
},
"id": 4,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_total_tokens_total[5m])) by (model)",
"legendFormat": "__auto",
"range": true,
"refId": "A"
}
],
"title": "Tokens",
"type": "timeseries"
}
],
"refresh": "1m",
"revision": 1,
"schemaVersion": 38,
"style": "dark",
"tags": [],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "prometheus",
"value": "edx8memhpd9tsa"
},
"hide": 0,
"includeAll": false,
"label": "datasource",
"multi": false,
"name": "DS_PROMETHEUS",
"options": [],
"query": "prometheus",
"queryValue": "",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
}
]
},
"time": {
"from": "now-1h",
"to": "now"
},
"timepicker": {},
"timezone": "",
"title": "LLM Proxy",
"uid": "rgRrHxESz",
"version": 15,
"weekStart": ""
}

View file

@ -1,6 +0,0 @@
## This folder contains the `json` for creating the following Grafana Dashboard
### Pre-Requisites
- Setup LiteLLM Proxy Prometheus Metrics https://docs.litellm.ai/docs/proxy/prometheus
![1716623265684](https://github.com/BerriAI/litellm/assets/29436595/0e12c57e-4a2d-4850-bd4f-e4294f87a814)

View file

@ -0,0 +1,11 @@
# LiteLLM All Prometheus Metrics dashboard
Every `litellm_*` metric family the proxy can expose on `/metrics` (134 families across 95 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
Import `grafana_dashboard.json` from **Dashboards > New > Import** and pick your Prometheus data source when prompted (the `DS_PROMETHEUS` variable). Counters are plotted as `rate()` over `$__rate_interval`, histograms as p50 / p95 / p99, gauges as the raw value grouped by the most useful label. Every query names the metric exactly as the proxy emits it (counters carry the `_total` suffix the Prometheus client adds), and `tests/test_litellm/integrations/test_prometheus_metric_name_consistency.py` fails if a metric is renamed without updating this dashboard
The first eleven rows need only `callbacks: ["prometheus"]`. The last three rows and the `litellm_admission_*` panels are emitted by other subsystems and stay empty until those are on: the service callback row needs `service_callback: ["prometheus_system"]` in `litellm_settings`, the circuit breaker row needs a Redis cache, the cleanup row needs spend log retention, and admission control needs its middleware enabled. Within the base rows, many panels only fill in once the matching feature is in use: budgets need keys, teams, users or orgs with `max_budget` set, cache panels need caching on, guardrail and MCP panels need those features configured, deployment health needs the router with more than one deployment or a failure to record, and `litellm_in_flight_requests` needs traffic at scrape time. An empty panel for a feature you do not use is expected
## Pre-requisites
Prometheus metrics on the proxy: https://docs.litellm.ai/docs/proxy/prometheus

View file

@ -476,7 +476,7 @@
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "topk(5, sort(litellm_remaining_requests))",
"expr": "topk(5, sort(litellm_remaining_requests_metric))",
"legendFormat": "__auto",
"range": true,
"refId": "A"
@ -573,7 +573,7 @@
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "topk(5, sort(litellm_remaining_tokens))",
"expr": "topk(5, sort(litellm_remaining_tokens_metric))",
"legendFormat": "__auto",
"range": true,
"refId": "A"

View file

@ -6,8 +6,14 @@ This folder contains the `json` for creating Grafana Dashboards
Charts the `gen_ai.*` metrics from the OpenTelemetry v2 integration: spend, tokens, request rate, and latency percentiles by model. Separate from the dashboards below, which chart the `litellm_*` Prometheus metrics.
## [LiteLLM All Prometheus Metrics dashboard](./dashboard_all_metrics)
Every `litellm_*` Prometheus metric family the proxy can emit (134 families, 95 panels) grouped by theme: traffic, latency, spend and tokens, cache, deployments, rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, plus the Redis circuit breaker, spend log cleanup and `prometheus_system` service metrics. Start here if you want everything on one screen; see its [readme](./dashboard_all_metrics/readme.md) for import steps and which panels need a feature enabled before they show data
## [LiteLLM v2 Dashboard](./dashboard_v2)
A compact view of proxy request rate, failures, latency and the top remaining-request / remaining-token gauges per model group
<img width="1316" alt="grafana_1" src="https://github.com/user-attachments/assets/d0df802d-0cb9-4906-a679-941c547789ab">
<img width="1289" alt="grafana_2" src="https://github.com/user-attachments/assets/b11f755f-e113-42ab-b21d-83f91f451a28">
<img width="1323" alt="grafana_3" src="https://github.com/user-attachments/assets/cb29ffdb-477d-4be1-a5cd-c3f7f2cb21c5">

View file

@ -96,6 +96,7 @@ GATEWAY_PATH_PREFIXES: tuple[str, ...] = (
"/langfuse/",
"/vllm/",
"/mistral/",
"/typesafe/",
"/nvidia_nim/",
"/groq/",
"/voyage/",

View file

@ -0,0 +1 @@
ALTER TABLE "LiteLLM_PolicyAttachmentTable" ADD COLUMN IF NOT EXISTS "priority" INTEGER;

View file

@ -1379,6 +1379,7 @@ model LiteLLM_PolicyAttachmentTable {
keys String[] @default([]) // Key aliases or patterns
models String[] @default([]) // Model names or patterns
tags String[] @default([]) // Tag patterns (e.g., ["healthcare", "prod-*"])
priority Int? // Explicit execution order
created_at DateTime @default(now())
created_by String?
updated_at DateTime @default(now()) @updatedAt

View file

@ -2016,6 +2016,7 @@ dependencies = [
"litellm-auth-azure",
"litellm-auth-gcp",
"litellm-framing",
"litellm-providers",
"mime_guess",
"moka",
"rand 0.8.7",
@ -2052,6 +2053,18 @@ dependencies = [
"tokio",
]
[[package]]
name = "litellm-providers"
version = "0.1.0"
dependencies = [
"litellm-auth",
"litellm-auth-aws",
"rstest",
"serde",
"serde_json",
"thiserror 2.0.19",
]
[[package]]
name = "litellm-python-bridge"
version = "0.1.0"

View file

@ -16,6 +16,7 @@ litellm-auth = { path = "crates/auth" }
litellm-auth-aws = { path = "crates/auth-aws" }
litellm-auth-azure = { path = "crates/auth-azure" }
litellm-auth-gcp = { path = "crates/auth-gcp" }
litellm-providers = { path = "crates/providers" }
litellm-cache = { path = "crates/cache" }
litellm-cache-memory = { path = "crates/cache-memory" }
litellm-token-counter = { path = "crates/token-counter" }

View file

@ -15,6 +15,7 @@ litellm-auth.workspace = true
litellm-auth-aws.workspace = true
litellm-auth-azure.workspace = true
litellm-auth-gcp.workspace = true
litellm-providers.workspace = true
litellm-framing.workspace = true
moka.workspace = true
mime_guess = "2.0.5"

View file

@ -24,3 +24,23 @@ pub enum Error {
#[error(transparent)]
Aws(#[from] litellm_auth_aws::Error),
}
impl From<litellm_providers::audio_transcription::Error> for Error {
fn from(error: litellm_providers::audio_transcription::Error) -> Self {
match error {
litellm_providers::audio_transcription::Error::InvalidType { expected, actual } => {
Self::InvalidType { expected, actual }
}
litellm_providers::audio_transcription::Error::MissingField(field) => {
Self::MissingField(field)
}
litellm_providers::audio_transcription::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::audio_transcription::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::audio_transcription::Error::Auth(error) => Self::Auth(error),
}
}
}

View file

@ -47,8 +47,8 @@ async fn signed_headers(
use std::collections::BTreeMap;
use std::time::SystemTime;
use crate::llms::base_llm::audio_transcription::transformation::AudioTranscriptionAuth;
use litellm_auth_aws::{aws_auth_config, resolve_credentials, sign_bedrock_post};
use litellm_providers::base_llm::audio_transcription::transformation::AudioTranscriptionAuth;
let AudioTranscriptionAuth::AwsSigV4 { region, .. } = &request.auth else {
return Ok(request.upstream_headers.clone());

View file

@ -3,7 +3,7 @@ pub use error::Error;
mod client;
mod handler;
mod prepare;
pub mod types;
pub use litellm_providers::audio_transcription::types;
pub use handler::execute_audio_transcription_provider_call;
pub use prepare::prepare_audio_transcription_provider_call;

View file

@ -4,10 +4,10 @@ use crate::http_utils::{has_header, string_headers};
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::audio_transcription::transformation::{
use litellm_providers::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
use crate::llms::bedrock::audio_transcription::BEDROCK_AUDIO_TRANSCRIPTION_CONFIG;
use litellm_providers::bedrock::audio_transcription::BEDROCK_AUDIO_TRANSCRIPTION_CONFIG;
fn provider_config(provider: &str) -> Option<&'static dyn BaseAudioTranscriptionConfig> {
if provider == "bedrock" {

View file

@ -2,8 +2,8 @@ use serde_json::{Map, Value};
use super::Error;
use crate::http_utils::string_headers as shared_string_headers;
use crate::llms::anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG;
use crate::llms::base_llm::chat::transformation::BaseConfig;
use litellm_providers::anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG;
use litellm_providers::base_llm::chat::transformation::BaseConfig;
const HEADER_CONTEXT: &str = "chat completions";
@ -11,7 +11,7 @@ pub(super) fn chat_completions_provider_config(provider: &str) -> Option<&'stati
match provider {
"anthropic" => Some(&ANTHROPIC_CHAT_COMPLETIONS_CONFIG),
"bedrock" => Some(
&crate::llms::bedrock::chat::converse_transformation::BEDROCK_CHAT_COMPLETIONS_CONFIG,
&litellm_providers::bedrock::chat::converse_transformation::BEDROCK_CHAT_COMPLETIONS_CONFIG,
),
_ => None,
}

View file

@ -24,3 +24,19 @@ pub enum Error {
#[error(transparent)]
Aws(#[from] litellm_auth_aws::Error),
}
impl From<litellm_providers::chat::Error> for Error {
fn from(error: litellm_providers::chat::Error) -> Self {
match error {
litellm_providers::chat::Error::MissingField(field) => Self::MissingField(field),
litellm_providers::chat::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::chat::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::chat::Error::Unsupported(reason) => Self::Unsupported(reason),
litellm_providers::chat::Error::Auth(error) => Self::Auth(error),
}
}
}

View file

@ -8,7 +8,7 @@ use super::types::{
ResolvedChatCompletionsRequest,
};
use crate::http_utils::{http_request, truncate_error_body};
use crate::llms::base_llm::chat::transformation::ChatCompletionsAuth;
use litellm_providers::base_llm::chat::transformation::ChatCompletionsAuth;
pub(super) async fn execute_chat_completions_provider_call(
request: ResolvedChatCompletionsRequest<'_>,
@ -59,6 +59,7 @@ pub(super) async fn execute_chat_completions_provider_call(
request
.config
.transform_response(&request.model, ProviderChatResponseData { body })
.map_err(Error::from)
.map_err(as_response_error)
}

View file

@ -10,12 +10,11 @@ mod error;
pub use error::Error;
mod client;
mod common_utils;
pub mod conversation;
pub use litellm_providers::chat::{conversation, response_utils};
pub(crate) mod handler;
mod prepare;
pub mod response_utils;
pub mod streaming;
pub mod types;
pub use litellm_providers::chat::types;
use handler::execute_chat_completions_provider_call;
use prepare::{parse_messages, resolve_provider_config, resolve_request};

View file

@ -10,7 +10,7 @@ use crate::http_utils::has_header;
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
use litellm_providers::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
pub(super) fn resolve_provider_config<'a>(
model: &'a str,

View file

@ -3,7 +3,7 @@ use serde_json::{Map, Value, json};
use super::Error;
use super::prepare::{prepare_provider_request, resolve_request};
use super::types::{ChatCompletionsRequest, ProviderChatCompletionsRequest};
use crate::llms::base_llm::chat::transformation::ChatCompletionsAuth;
use litellm_providers::base_llm::chat::transformation::ChatCompletionsAuth;
fn prepare_chat_completions_call(
request: ChatCompletionsRequest<'_>,

View file

@ -1,36 +1,4 @@
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct CustomLlmProvider<'a> {
pub model: &'a str,
pub custom_llm_provider: &'a str,
}
pub fn get_custom_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
) -> Option<CustomLlmProvider<'a>> {
if let Some(custom_llm_provider) = custom_llm_provider.filter(|provider| !provider.is_empty()) {
return Some(CustomLlmProvider {
model: strip_custom_llm_provider_prefix(model, custom_llm_provider),
custom_llm_provider,
});
}
let (custom_llm_provider, model) = model.split_once('/')?;
if custom_llm_provider.is_empty() || model.is_empty() {
return None;
}
Some(CustomLlmProvider {
model,
custom_llm_provider,
})
}
fn strip_custom_llm_provider_prefix<'a>(model: &'a str, custom_llm_provider: &str) -> &'a str {
model
.strip_prefix(custom_llm_provider)
.and_then(|model| model.strip_prefix('/'))
.unwrap_or(model)
}
pub use litellm_providers::provider_resolution::{CustomLlmProvider, get_custom_llm_provider};
#[cfg(test)]
mod tests {

View file

@ -1,2 +1 @@
pub mod streaming;
pub mod transformation;

View file

@ -2,16 +2,16 @@ use std::collections::HashMap;
use serde_json::Value;
use super::super::experimental_pass_through::messages::streaming::{
AnthropicContentBlock, AnthropicContentBlockDelta, AnthropicMessagesStreamEvent,
AnthropicStreamUsage,
};
use crate::chat_completions::Error;
use crate::chat_completions::streaming::StreamTransformer;
use crate::chat_completions::types::{
ChatCompletionChunk, ChatCompletionThinkingBlock, ChatCompletionToolCallChunk,
ChatCompletionsUsage,
};
use crate::llms::anthropic::experimental_pass_through::messages::streaming::{
AnthropicContentBlock, AnthropicContentBlockDelta, AnthropicMessagesStreamEvent,
AnthropicStreamUsage,
};
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum AnthropicJsonChunkType {

View file

@ -3,9 +3,9 @@ use serde_json::Value;
use time::OffsetDateTime;
use url::Url;
use crate::llms::anthropic::experimental_pass_through::messages::transformation::resolve_anthropic_api_base;
use crate::messages::Error;
use crate::messages::types::AnthropicMessagesResponse;
use litellm_providers::anthropic::experimental_pass_through::messages::transformation::resolve_anthropic_api_base;
const BATCHES_PATH_SUFFIX: &str = "/v1/messages/batches";

View file

@ -1,4 +1,3 @@
pub mod batches;
pub mod count_tokens;
pub mod streaming;
pub mod transformation;

View file

@ -1,2 +1 @@
pub mod anthropic;
pub(crate) mod ocr;

View file

@ -1,4 +1 @@
pub mod anthropic_messages;
pub mod audio_transcription;
pub mod chat;
pub(crate) mod ocr;

View file

@ -1,7 +1,6 @@
pub mod anthropic;
pub mod azure_ai;
pub mod base_llm;
pub mod bedrock;
pub(crate) mod cohere;
pub(crate) mod mistral;
pub mod openai;

View file

@ -3,9 +3,9 @@ use serde_json::{Map, Value};
use super::Error;
use crate::http_utils::string_headers as shared_string_headers;
pub(super) use crate::http_utils::{has_bearer_auth, has_header, truncate_error_body};
use crate::llms::anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG;
use crate::llms::azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG;
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use litellm_providers::anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG;
use litellm_providers::azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG;
use litellm_providers::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
const HEADER_CONTEXT: &str = "messages";

View file

@ -28,6 +28,22 @@ pub enum Error {
InvalidBedrockBase64(String),
}
impl From<litellm_providers::messages::Error> for Error {
fn from(error: litellm_providers::messages::Error) -> Self {
match error {
litellm_providers::messages::Error::MissingField(field) => Self::MissingField(field),
litellm_providers::messages::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::messages::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::messages::Error::Unsupported(reason) => Self::Unsupported(reason),
litellm_providers::messages::Error::Auth(error) => Self::Auth(error),
}
}
}
impl Error {
pub fn is_request(&self) -> bool {
match self {

View file

@ -40,6 +40,7 @@ pub(super) async fn execute_messages_provider_call(
request
.config
.transform_anthropic_messages_response(&request.model, response)
.map_err(Error::from)
}
pub(super) async fn execute_messages_provider_stream(

View file

@ -13,7 +13,7 @@ mod client;
mod common_utils;
mod handler;
mod prepare;
pub mod types;
pub use litellm_providers::messages::types;
use handler::{execute_messages_provider_call, execute_messages_provider_stream};
use types::{AnthropicMessagesResponse, MessagesRequest};

View file

@ -6,7 +6,7 @@ use super::types::{MessagesRequest, ProviderMessagesRequest};
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::anthropic_messages::transformation::{
use litellm_providers::base_llm::anthropic_messages::transformation::{
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
};

View file

@ -0,0 +1,16 @@
[package]
name = "litellm-providers"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
litellm-auth.workspace = true
litellm-auth-aws.workspace = true
serde.workspace = true
serde_json.workspace = true
thiserror.workspace = true
[dev-dependencies]
rstest.workspace = true

View file

@ -1,7 +1,7 @@
use serde_json::json;
use super::*;
use crate::chat_completions::Error;
use crate::chat::Error;
fn messages(value: Value) -> Vec<ChatMessage> {
serde_json::from_value(value).expect("valid messages")

View file

@ -1,19 +1,19 @@
use serde_json::{Map, Value, json};
use crate::chat_completions::Error;
use crate::chat_completions::conversation::{Conversation, build_conversation};
use crate::chat_completions::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat_completions::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
ProviderChatRequestData, ProviderChatResponseData,
};
use crate::constants::ANTHROPIC_OAUTH_TOKEN_PREFIX;
use crate::llms::anthropic::experimental_pass_through::messages::transformation::{
use crate::anthropic::ANTHROPIC_OAUTH_TOKEN_PREFIX;
use crate::anthropic::experimental_pass_through::messages::transformation::{
complete_anthropic_url, resolve_anthropic_api_key,
};
use crate::llms::base_llm::chat::transformation::{
use crate::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use crate::chat::Error;
use crate::chat::conversation::{Conversation, build_conversation};
use crate::chat::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
ProviderChatRequestData, ProviderChatResponseData,
};
/// Anthropic parameter names, post `map_openai_params`, that the Rust path can
/// place verbatim in the Messages body.

View file

@ -1,4 +1,4 @@
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::messages::Error;
const ANTHROPIC_API_KEY_ENV: &str = "ANTHROPIC_API_KEY";

View file

@ -0,0 +1 @@
pub mod messages;

View file

@ -0,0 +1,4 @@
pub mod chat;
pub mod experimental_pass_through;
pub const ANTHROPIC_OAUTH_TOKEN_PREFIX: &str = "sk-ant-oat";

View file

@ -0,0 +1,31 @@
use thiserror::Error;
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("expected {expected}, got {actual}")]
InvalidType {
expected: &'static str,
actual: &'static str,
},
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub fn json_type_name(value: &serde_json::Value) -> &'static str {
match value {
serde_json::Value::Null => "null",
serde_json::Value::Bool(_) => "boolean",
serde_json::Value::Number(_) => "number",
serde_json::Value::String(_) => "string",
serde_json::Value::Array(_) => "array",
serde_json::Value::Object(_) => "object",
}
}
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::audio_transcription::transformation::{
use crate::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
@ -20,15 +20,15 @@ pub struct AudioTranscriptionRequest<'a> {
#[derive(Clone)]
pub struct ProviderAudioTranscriptionRequest {
pub(super) model: String,
pub(super) custom_llm_provider: String,
pub(super) config: &'static dyn BaseAudioTranscriptionConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) auth: AudioTranscriptionAuth,
pub(super) optional_params: Map<String, Value>,
pub(super) timeout: Option<Duration>,
pub model: String,
pub custom_llm_provider: String,
pub config: &'static dyn BaseAudioTranscriptionConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub auth: AudioTranscriptionAuth,
pub optional_params: Map<String, Value>,
pub timeout: Option<Duration>,
}
impl ProviderAudioTranscriptionRequest {

View file

@ -1,9 +1,9 @@
use serde_json::{Map, Value};
use crate::llms::anthropic::experimental_pass_through::messages::transformation::{
use crate::anthropic::experimental_pass_through::messages::transformation::{
ANTHROPIC_MESSAGES_CONFIG, AnthropicMessagesConfig, non_empty,
};
use crate::llms::base_llm::anthropic_messages::transformation::{
use crate::base_llm::anthropic_messages::transformation::{
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
};
use crate::messages::Error;

View file

@ -0,0 +1 @@
pub mod anthropic;

View file

@ -0,0 +1 @@
pub mod transformation;

View file

@ -0,0 +1 @@
pub mod transformation;

View file

@ -1,7 +1,7 @@
use serde_json::{Map, Value};
use crate::chat_completions::Error;
use crate::chat_completions::types::{
use crate::chat::Error;
use crate::chat::types::{
ChatCompletionsResponse, ChatMessage, ChatMessageContent, ProviderChatRequestData,
ProviderChatResponseData,
};

View file

@ -0,0 +1,3 @@
pub mod anthropic_messages;
pub mod audio_transcription;
pub mod chat;

View file

@ -1,11 +1,11 @@
use serde_json::{Map, Value, json};
use crate::audio_transcription::Error;
use crate::audio_transcription::json_type_name;
use crate::audio_transcription::types::{
AudioTranscriptionRequestData, AudioTranscriptionResponseData,
};
use crate::http_utils::json_type_name;
use crate::llms::base_llm::audio_transcription::transformation::{
use crate::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
use litellm_auth_aws::constants::{BEDROCK_RUNTIME_ENDPOINT_TEMPLATE, BEDROCK_SERVICE};

View file

@ -1,16 +1,16 @@
use serde_json::{Map, Value, json};
use crate::chat_completions::Error;
use crate::chat_completions::conversation::{Conversation, TurnRole, build_conversation};
use crate::chat_completions::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat_completions::types::{
use crate::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use crate::chat::Error;
use crate::chat::conversation::{Conversation, TurnRole, build_conversation};
use crate::chat::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse,
ChatCompletionsUsage, ChatMessage, ChatMessageContent, ProviderChatRequestData,
ProviderChatResponseData,
};
use crate::llms::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use litellm_auth_aws::constants::{AWS_BEARER_TOKEN_BEDROCK, BEDROCK_RUNTIME_ENDPOINT_TEMPLATE};
use litellm_auth_aws::{bedrock_model_id_and_region, resolve_bedrock_region};

View file

@ -1,7 +1,7 @@
use serde_json::json;
use super::*;
use crate::chat_completions::Error;
use crate::chat::Error;
fn messages(value: Value) -> Vec<ChatMessage> {
serde_json::from_value(value).expect("valid messages")

View file

@ -11,7 +11,7 @@
//! accepts; anything richer is declined upstream by the capability gate.
use super::types::{ChatMessage, ChatMessageContent};
use crate::constants::EMPTY_TEXT_PLACEHOLDER;
use crate::chat::EMPTY_TEXT_PLACEHOLDER;
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum TurnRole {

View file

@ -0,0 +1,21 @@
use thiserror::Error;
pub const EMPTY_TEXT_PLACEHOLDER: &str = " ";
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error("unsupported: {0}")]
Unsupported(&'static str),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub mod conversation;
pub mod response_utils;
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
use crate::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
/// A `/chat/completions` call as it crosses into the core.
///
@ -22,26 +22,26 @@ pub struct ChatCompletionsRequest<'a> {
pub timeout: Option<Duration>,
}
pub(super) struct ResolvedChatCompletionsRequest<'a> {
pub(super) model: String,
pub(super) config: &'static dyn BaseConfig,
pub(super) messages: Vec<ChatMessage>,
pub(super) optional_params: Map<String, Value>,
pub(super) api_key: Option<&'a str>,
pub(super) api_base: Option<&'a str>,
pub(super) extra_headers: Option<Map<String, Value>>,
pub(super) timeout: Option<Duration>,
pub struct ResolvedChatCompletionsRequest<'a> {
pub model: String,
pub config: &'static dyn BaseConfig,
pub messages: Vec<ChatMessage>,
pub optional_params: Map<String, Value>,
pub api_key: Option<&'a str>,
pub api_base: Option<&'a str>,
pub extra_headers: Option<Map<String, Value>>,
pub timeout: Option<Duration>,
}
pub(super) struct ProviderChatCompletionsRequest {
pub(super) model: String,
pub(super) config: &'static dyn BaseConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) auth: ChatCompletionsAuth,
pub(super) optional_params: Map<String, Value>,
pub(super) timeout: Option<Duration>,
pub struct ProviderChatCompletionsRequest {
pub model: String,
pub config: &'static dyn BaseConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub auth: ChatCompletionsAuth,
pub optional_params: Map<String, Value>,
pub timeout: Option<Duration>,
}
/// The provider-shaped request body a config produces. Named rather than a bare

View file

@ -0,0 +1,8 @@
pub mod anthropic;
pub mod audio_transcription;
pub mod azure_ai;
pub mod base_llm;
pub mod bedrock;
pub mod chat;
pub mod messages;
pub mod provider_resolution;

View file

@ -0,0 +1,17 @@
use thiserror::Error;
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error("unsupported: {0}")]
Unsupported(&'static str),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
pub struct MessagesRequest<'a> {
pub model: &'a str,
@ -15,14 +15,14 @@ pub struct MessagesRequest<'a> {
pub timeout: Option<Duration>,
}
pub(super) struct ProviderMessagesRequest {
pub(super) provider: String,
pub(super) model: String,
pub(super) config: &'static dyn BaseAnthropicMessagesConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) timeout: Option<Duration>,
pub struct ProviderMessagesRequest {
pub provider: String,
pub model: String,
pub config: &'static dyn BaseAnthropicMessagesConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub timeout: Option<Duration>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]

View file

@ -0,0 +1,33 @@
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct CustomLlmProvider<'a> {
pub model: &'a str,
pub custom_llm_provider: &'a str,
}
pub fn get_custom_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
) -> Option<CustomLlmProvider<'a>> {
if let Some(custom_llm_provider) = custom_llm_provider.filter(|provider| !provider.is_empty()) {
return Some(CustomLlmProvider {
model: strip_custom_llm_provider_prefix(model, custom_llm_provider),
custom_llm_provider,
});
}
let (custom_llm_provider, model) = model.split_once('/')?;
if custom_llm_provider.is_empty() || model.is_empty() {
return None;
}
Some(CustomLlmProvider {
model,
custom_llm_provider,
})
}
fn strip_custom_llm_provider_prefix<'a>(model: &'a str, custom_llm_provider: &str) -> &'a str {
model
.strip_prefix(custom_llm_provider)
.and_then(|model| model.strip_prefix('/'))
.unwrap_or(model)
}

View file

@ -320,6 +320,8 @@ WEBSOCKET_CLOSE_REASON_MAX_BYTES: Final = 123
BEDROCK_REALTIME_PENDING_SESSION_UPDATE_SCOPE_KEY: Final = "litellm.bedrock_realtime.pending_session_update"
BEDROCK_REALTIME_SESSION_COMMITTED_SCOPE_KEY: Final = "litellm.bedrock_realtime.session_committed"
BEDROCK_REALTIME_COMMITTED_FAILURE_SCOPE_KEY: Final = "litellm.bedrock_realtime.committed_failure"
BEDROCK_REALTIME_SDK_DISTRIBUTION: Final = "aws-sdk-bedrock-runtime"
BEDROCK_REALTIME_SDK_SUPPORTED_RANGE: Final = ">=0.10.0,<0.12.0"
CLIENT_REQUESTED_MODEL_SCOPE_KEY: Final = "litellm.client_requested_model"
MODEL_GROUP_ALIAS_RESOLVED_SCOPE_KEY: Final = "litellm.model_group_alias_resolved"
REALTIME_SESSION_SUCCESS_LOGGED_KEY: Final = "realtime_session_success_logged"

View file

@ -90,6 +90,7 @@ from litellm.litellm_core_utils.logging_utils import (
truncate_base64_in_messages_async,
)
from litellm.litellm_core_utils.model_param_helper import ModelParamHelper
from litellm.litellm_core_utils.ptu_pricing import is_spilled_over_ptu_request
from litellm.litellm_core_utils.redact_messages import (
redact_message_input_output_from_custom_logger,
redact_message_input_output_from_logging,
@ -1746,8 +1747,14 @@ class Logging(LiteLLMLoggingBaseClass):
if transformed_result is not None:
result = transformed_result
result_hidden_params: Final = getattr(result, "_hidden_params", None) or MappingProxyType({})
result_additional_headers: Final = (
result_hidden_params.get("additional_headers")
if isinstance(result_hidden_params, dict)
else getattr(result_hidden_params, "additional_headers", None)
)
if isinstance(result, (BaseModel, HttpxBinaryResponseContent)) and hasattr(result, "_hidden_params"):
hidden_params: Final = getattr(result, "_hidden_params", {})
hidden_params: Final = result_hidden_params
if (
"response_cost" in hidden_params and hidden_params["response_cost"] is not None
): # use cost if already calculated
@ -1762,8 +1769,17 @@ class Logging(LiteLLMLoggingBaseClass):
router_model_id = self.get_router_model_id()
## RESPONSE COST ##
custom_pricing: Final = use_custom_pricing_for_model(
litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None)
spilled_over: Final = is_spilled_over_ptu_request(
model_info=_deployment_model_info(self.litellm_params if hasattr(self, "litellm_params") else None),
response_headers=self.model_call_details.get("response_headers"),
additional_headers=result_additional_headers,
)
custom_pricing: Final = (
False
if spilled_over
else use_custom_pricing_for_model(
litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None)
)
)
prompt = self._prompt_for_cost_calculation()
@ -5257,6 +5273,18 @@ def _get_custom_logger_settings_from_proxy_server(callback_name: str) -> dict:
return {}
def _deployment_model_info(litellm_params: dict | None) -> Mapping[str, object]:
"""The router-stamped deployment model_info from whichever metadata field carries it."""
if litellm_params is None:
return MappingProxyType({})
for metadata_key in ("metadata", "litellm_metadata"):
if not isinstance(metadata := litellm_params.get(metadata_key), Mapping):
continue
if model_info := metadata.get("model_info"):
return model_info
return MappingProxyType({})
def use_custom_pricing_for_model(litellm_params: dict | None) -> bool:
"""
Check if the model uses custom pricing

View file

@ -14,9 +14,11 @@ from typing import Final
from litellm.secret_managers.main import get_secret_bool
from litellm.types.router import ModelInfo
from litellm.types.utils import CustomPricingLiteLLMParams, MirroredPricingParams
from litellm.types.utils import AzureSpillover, CustomPricingLiteLLMParams, MirroredPricingParams
PTU_COST_ATTRIBUTION_ENV_VAR: Final = "LITELLM_ENABLE_PTU_COST_ATTRIBUTION"
AZURE_SPILLOVER_HEADER: Final = "x-ms-is-spilled-over"
AZURE_SPILLOVER_FROM_HEADER: Final = "x-ms-spillover-from-deployment"
def is_ptu_cost_attribution_enabled() -> bool:
@ -235,3 +237,33 @@ def zeroed_ptu_pricing(
),
}
)
def is_spilled_over_ptu_request(
model_info: Mapping[str, object],
response_headers: Mapping[str, object] | None,
additional_headers: Mapping[str, object] | None,
) -> bool:
"""Whether Azure served this request from pay-as-you-go capacity, so the zeroed PTU rates must not apply."""
if ptu_terms(model_info) is None:
return False
if not is_ptu_cost_attribution_enabled():
return False
return azure_spillover(response_headers, additional_headers) is not None
def azure_spillover(
response_headers: Mapping[str, object] | None,
additional_headers: Mapping[str, object] | None,
) -> AzureSpillover | None:
"""The spillover Azure reports in the response headers, else None."""
for headers, prefix in (
(response_headers, ""),
(additional_headers, "llm_provider-"),
):
if headers is None or str(headers.get(f"{prefix}{AZURE_SPILLOVER_HEADER}")).lower() != "true":
continue
return AzureSpillover(
from_deployment=str(v) if (v := headers.get(f"{prefix}{AZURE_SPILLOVER_FROM_HEADER}")) is not None else None
)
return None

View file

@ -508,7 +508,8 @@ class AnthropicMessagesHandler(BaseTranslation):
chat_completion_compatible_request,
_tool_name_mapping,
) = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request=cast(AnthropicMessagesRequest, data.copy())
anthropic_message_request=cast(AnthropicMessagesRequest, data.copy()),
preserve_midturn_system=True,
)
return chat_completion_compatible_request

View file

@ -118,6 +118,10 @@ from litellm.llms.anthropic.common_utils import (
from litellm.llms.anthropic.experimental_pass_through.context_management import (
PolyfillResult,
)
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
convert_mid_conversation_system_turns,
is_system_role_message,
)
from litellm.llms.anthropic.experimental_pass_through.messages.utils import (
openai_chat_refusal_text,
refusal_stop_details,
@ -176,6 +180,7 @@ from litellm.types.llms.openai import (
ToolMessageContentPart,
)
from litellm.types.utils import Choices, ModelResponse, StreamingChoices, Usage
from litellm.utils import supports_mid_conversation_system
from .streaming_iterator import AnthropicStreamWrapper
@ -186,6 +191,12 @@ if TYPE_CHECKING:
ToolResultContent: TypeAlias = str | list[ToolMessageContentPart]
def target_supports_mid_conversation_system(model: str | None, custom_llm_provider: str | None) -> bool:
if not model:
return False
return supports_mid_conversation_system(model=model, custom_llm_provider=custom_llm_provider)
class AnthropicAdapter:
def __init__(self) -> None:
pass
@ -418,10 +429,28 @@ class LiteLLMAnthropicMessagesAdapter:
self,
messages: list[AllAnthropicPassThroughMessageValues],
model: str | None = None,
*,
custom_llm_provider: str | None = None,
preserve_midturn_system: bool = False,
) -> list:
new_messages: Final[list[AllMessageValues]] = []
replayable_messages: Final = strip_encrypted_reasoning_blocks_from_anthropic_messages(messages)
for m in replayable_messages:
leading_count: Final = next(
(i for i, m in enumerate(replayable_messages) if not is_system_role_message(m)),
len(replayable_messages),
)
trailing_messages: Final = replayable_messages[leading_count:]
keeps_midturn_system: Final = (
preserve_midturn_system
or not any(is_system_role_message(m) for m in trailing_messages)
or target_supports_mid_conversation_system(model, custom_llm_provider)
)
ordered_messages: Final = (
replayable_messages
if keeps_midturn_system
else (*replayable_messages[:leading_count], *convert_mid_conversation_system_turns(trailing_messages))
)
for m in ordered_messages:
user_message: ChatCompletionUserMessage | None = None
tool_message_list: list[ChatCompletionToolMessage] = []
new_user_content_list: list[ChatCompletionTextObject | ChatCompletionImageObject] = []
@ -494,7 +523,7 @@ class LiteLLMAnthropicMessagesAdapter:
if isinstance(m.get("content"), str):
assistant_message_str = str(m.get("content", ""))
elif isinstance(m.get("content"), list):
for content in m.get("content", []):
for content in cast(list, m.get("content", [])): # cast-ok: untrusted client payload
if isinstance(content, str):
assistant_message_str = str(content)
elif isinstance(content, dict):
@ -1154,6 +1183,7 @@ class LiteLLMAnthropicMessagesAdapter:
anthropic_message_request: AnthropicMessagesRequest,
*,
custom_llm_provider: str | None = None,
preserve_midturn_system: bool = False,
) -> tuple[ChatCompletionRequest, dict[str, str]]:
"""
This is used by the beta Anthropic Adapter, for translating anthropic `/v1/messages` requests to the openai format.
@ -1175,6 +1205,8 @@ class LiteLLMAnthropicMessagesAdapter:
new_messages = self.translate_anthropic_messages_to_openai(
messages=messages_list,
model=anthropic_message_request.get("model"),
custom_llm_provider=custom_llm_provider,
preserve_midturn_system=preserve_midturn_system,
)
## ADD SYSTEM MESSAGE TO MESSAGES
self._add_system_message_to_messages(new_messages, anthropic_message_request)

View file

@ -765,7 +765,8 @@ def _count_effective_tokens(
messages=cast(
"list[AllAnthropicPassThroughMessageValues]",
messages_without_compaction,
)
),
preserve_midturn_system=True,
)
except Exception as e:
verbose_logger.debug(
@ -920,7 +921,8 @@ def _build_summary_messages(
messages=cast(
"list[AllAnthropicPassThroughMessageValues]",
stripped,
)
),
preserve_midturn_system=True,
)
except Exception as e:
verbose_logger.warning(

View file

@ -0,0 +1,77 @@
from collections.abc import Mapping, Sequence
from itertools import groupby
from typing import Final
CONVERTED_SYSTEM_NOTE: Final = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
def as_system_content_blocks(value: object) -> list[object]:
if value is None:
return []
if isinstance(value, list):
return list(value)
if isinstance(value, str):
return [{"type": "text", "text": value}]
return [value]
def is_system_role_message(message: object) -> bool:
return isinstance(message, dict) and message.get("role") == "system"
def system_role_message_as_user(message: Mapping[str, object]) -> Mapping[str, object]:
return {
"role": "user",
"content": as_system_content_blocks(CONVERTED_SYSTEM_NOTE) + as_system_content_blocks(message.get("content")),
}
def opens_with_tool_results(message: object) -> bool:
if not isinstance(message, dict) or message.get("role") != "user":
return False
content: Final = message.get("content")
return (
isinstance(content, list)
and len(content) > 0
and isinstance(content[0], dict)
and content[0].get("type") == "tool_result"
)
def system_run_placed_after_tool_results(
system_run: Sequence[Mapping[str, object]], follower_run: Sequence[Mapping[str, object]]
) -> tuple[Mapping[str, object], ...]:
if follower_run and opens_with_tool_results(follower_run[0]):
return (follower_run[0], *system_run, *follower_run[1:])
return (*system_run, *follower_run)
def system_turns_after_tool_results(
messages: Sequence[Mapping[str, object]],
) -> tuple[Mapping[str, object], ...]:
runs: Final = tuple(tuple(run) for _, run in groupby(messages, key=is_system_role_message))
if not runs:
return ()
first_system_run: Final = 0 if is_system_role_message(runs[0][0]) else 1
paired_runs: Final = tuple(
(runs[i], runs[i + 1] if i + 1 < len(runs) else ()) for i in range(first_system_run, len(runs), 2)
)
return (
*(runs[0] if first_system_run else ()),
*(
m
for system_run, follower_run in paired_runs
for m in system_run_placed_after_tool_results(system_run, follower_run)
),
)
def convert_mid_conversation_system_turns(
messages: Sequence[Mapping[str, object]],
) -> tuple[Mapping[str, object], ...]:
return tuple(
system_role_message_as_user(m) if is_system_role_message(m) else m
for m in system_turns_after_tool_results(messages)
)

View file

@ -27,6 +27,11 @@ from ...common_utils import (
strip_advisor_blocks_from_messages,
strip_encrypted_reasoning_blocks_from_anthropic_messages,
)
from .mid_conversation_system import (
as_system_content_blocks,
convert_mid_conversation_system_turns,
is_system_role_message,
)
DEFAULT_ANTHROPIC_API_VERSION: Final = "2023-06-01"
@ -151,73 +156,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
else:
return system_param
@staticmethod
def _as_system_content_blocks(value: object) -> list:
if value is None:
return []
if isinstance(value, list):
return list(value)
if isinstance(value, str):
return [{"type": "text", "text": value}]
return [value]
@staticmethod
def _is_system_role_message(message: object) -> bool:
return isinstance(message, dict) and message.get("role") == "system"
_CONVERTED_SYSTEM_NOTE: Final = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
def _system_role_message_as_user(self, message: Mapping) -> Mapping:
return {
"role": "user",
"content": self._as_system_content_blocks(self._CONVERTED_SYSTEM_NOTE)
+ self._as_system_content_blocks(message.get("content")),
}
@staticmethod
def _opens_with_tool_results(message: object) -> bool:
if not isinstance(message, dict) or message.get("role") != "user":
return False
content: Final = message.get("content")
return (
isinstance(content, list)
and len(content) > 0
and isinstance(content[0], dict)
and content[0].get("type") == "tool_result"
)
def _system_run_before(self, messages: Sequence, index: int) -> Sequence:
start: Final = next(
(j + 1 for j in range(index - 1, -1, -1) if not self._is_system_role_message(messages[j])),
0,
)
return messages[start:index]
def _system_run_end(self, messages: Sequence, index: int) -> int:
return next(
(j for j in range(index, len(messages)) if not self._is_system_role_message(messages[j])),
len(messages),
)
def _reordered_around_tool_results(self, messages: Sequence, index: int) -> tuple:
message: Final = messages[index]
if self._opens_with_tool_results(message):
return (message, *self._system_run_before(messages, index))
if not self._is_system_role_message(message):
return (message,)
run_end: Final = self._system_run_end(messages, index)
follower: Final = messages[run_end] if run_end < len(messages) else None
return () if self._opens_with_tool_results(follower) else (message,)
def _system_turns_after_tool_results(self, messages: Sequence) -> tuple:
return tuple(
message
for index in range(len(messages))
for message in self._reordered_around_tool_results(messages, index)
)
def _normalize_system_role_messages(self, anthropic_messages_request: dict, model: str) -> None:
"""Normalize ``role: "system"`` entries in ``messages`` per the Anthropic
``/v1/messages`` contract, which the first-party API, Bedrock Invoke,
@ -254,7 +192,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
if not isinstance(messages, list):
return
leading_count: Final = next(
(i for i, m in enumerate(messages) if not self._is_system_role_message(m)),
(i for i, m in enumerate(messages) if not is_system_role_message(m)),
len(messages),
)
hoisted: Final = messages[:leading_count]
@ -265,10 +203,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
custom_llm_provider=self.custom_llm_provider,
key="supports_mid_conversation_system",
)
else [
self._system_role_message_as_user(m) if self._is_system_role_message(m) else m
for m in self._system_turns_after_tool_results(messages[leading_count:])
]
else list(convert_mid_conversation_system_turns(messages[leading_count:]))
)
if hoisted or remaining != messages:
anthropic_messages_request["messages"] = remaining
@ -278,7 +213,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
anthropic_messages_request.get("system"),
*(m.get("content") for m in hoisted),
)
for block in self._as_system_content_blocks(source)
for block in as_system_content_blocks(source)
]
filtered_system: Final = self._filter_billing_headers_from_system(system_content)
if filtered_system:

View file

@ -561,6 +561,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM):
headers, response = self.make_sync_azure_openai_chat_completion_request(
azure_client=azure_client, data=data, timeout=timeout
)
logging_obj.model_call_details["response_headers"] = headers
streamwrapper: Final = CustomStreamWrapper(
completion_stream=response,
model=model,

View file

@ -34,6 +34,7 @@ if TYPE_CHECKING:
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
def error_response_text(response: httpx.Response) -> str:
@ -878,9 +879,10 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
Whether Converse ``cachePoint`` blocks may be sent to this model.
Bedrock rejects requests carrying cachePoint blocks for models without prompt
caching support ("You invoked an unsupported model or your request did not allow
prompt caching"), so a model whose cost-map entry does not declare
OpenAI-family models only support implicit caching and never accept explicit
``cachePoint`` blocks. Bedrock rejects requests carrying cachePoint blocks for
models without prompt caching support ("You invoked an unsupported model or your
request did not allow prompt caching"), so a model whose cost-map entry does not declare
``supports_prompt_caching`` must not receive them. A model absent from the map
(an application inference profile ARN, a model newer than the map) keeps emitting
so existing caching setups never silently degrade. ``litellm.utils.supports_prompt_caching``
@ -888,6 +890,8 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
if model is None:
return True
if _OPENAI_FAMILY_MODEL_RE.search(model):
return False
entries: Final = tuple(
entry
for candidate in (model, get_bedrock_base_model(model))

View file

@ -6,11 +6,12 @@ This uses aws_sdk_bedrock_runtime for bidirectional streaming with Nova Sonic.
import asyncio
import contextlib
import importlib.metadata
import json
from collections.abc import AsyncIterator, Mapping, MutableMapping
from collections.abc import AsyncIterator, Awaitable, Callable, Mapping, MutableMapping
from dataclasses import dataclass
from types import MappingProxyType
from typing import Final, NoReturn, Protocol
from typing import Final, NoReturn, Protocol, runtime_checkable
from pydantic import JsonValue, TypeAdapter
@ -19,6 +20,8 @@ from litellm._logging import _redact_string, verbose_proxy_logger
from litellm.constants import (
BEDROCK_REALTIME_COMMITTED_FAILURE_SCOPE_KEY,
BEDROCK_REALTIME_PENDING_SESSION_UPDATE_SCOPE_KEY,
BEDROCK_REALTIME_SDK_DISTRIBUTION,
BEDROCK_REALTIME_SDK_SUPPORTED_RANGE,
BEDROCK_REALTIME_SESSION_COMMITTED_SCOPE_KEY,
REALTIME_SESSION_SUCCESS_LOGGED_KEY,
)
@ -121,6 +124,38 @@ class BedrockBidirectionalStream(Protocol):
async def await_output(self) -> tuple[object, BedrockOutputStream]: ...
@runtime_checkable
class ClosableBedrockRuntimeClient(Protocol):
async def close(self) -> None: ...
def _installed_sdk_version() -> str | None:
try:
return importlib.metadata.version(BEDROCK_REALTIME_SDK_DISTRIBUTION)
except importlib.metadata.PackageNotFoundError:
return None
def _sdk_import_error(installed_version: str | None, cause: ImportError) -> ImportError:
install_hint: Final = "pip install 'litellm[bedrock-realtime]'"
requirement: Final = f"{BEDROCK_REALTIME_SDK_DISTRIBUTION}[awscrt]{BEDROCK_REALTIME_SDK_SUPPORTED_RANGE}"
verbose_proxy_logger.error("Bedrock Realtime: SDK import failed (installed=%s): %s", installed_version, cause)
if installed_version is None:
return ImportError(f"Missing aws_sdk_bedrock_runtime: {install_hint} ({requirement})")
return ImportError(
f"{BEDROCK_REALTIME_SDK_DISTRIBUTION} {installed_version} is installed but Bedrock realtime needs "
f"[awscrt]{BEDROCK_REALTIME_SDK_SUPPORTED_RANGE}: {install_hint}"
)
async def _close_bedrock_client(bedrock_client: object) -> None:
if not isinstance(bedrock_client, ClosableBedrockRuntimeClient):
return
with contextlib.suppress(Exception):
await bedrock_client.close()
verbose_proxy_logger.debug("Bedrock Realtime: closed SDK client")
@dataclass(frozen=True, slots=True)
class _BridgeOutcome:
logged_events: tuple[OpenAIRealtimeEvents, ...]
@ -199,8 +234,9 @@ async def _ack_session_update(
class BedrockRealtime(BaseAWSLLM):
"""Handler for Bedrock Nova Sonic realtime speech-to-speech API."""
def __init__(self):
def __init__(self, sdk_version_lookup: Callable[[], str | None] = _installed_sdk_version):
super().__init__()
self._sdk_version_lookup: Final = sdk_version_lookup
async def async_realtime(
self,
@ -234,14 +270,13 @@ class BedrockRealtime(BaseAWSLLM):
Various AWS authentication parameters
"""
try:
from aws_sdk_bedrock_runtime.client import (
BedrockRuntimeClient,
InvokeModelWithBidirectionalStreamOperationInput,
)
from aws_sdk_bedrock_runtime.config import Config
from smithy_aws_core.identity import StaticCredentialsResolver
except ImportError:
raise ImportError("Missing aws_sdk_bedrock_runtime. Install with: pip install aws-sdk-bedrock-runtime")
from aws_sdk_bedrock_runtime.client import AsyncBedrockRuntimeClient
from aws_sdk_bedrock_runtime.config import AsyncBedrockRuntimeConfig
from aws_sdk_bedrock_runtime.models import InvokeModelWithBidirectionalStreamOperationInput
from smithy_aws_core.identity import AWSCredentialsIdentity, StaticCredentialsResolver
from smithy_http.aio.crt import AWSCRTHTTPClient
except ImportError as e:
raise _sdk_import_error(self._sdk_version_lookup(), e) from e
pending_session_update: Final = _pending_session_update(websocket.scope)
@ -285,22 +320,37 @@ class BedrockRealtime(BaseAWSLLM):
)
frozen_credentials: Final = await run_aws_signing(credentials.get_frozen_credentials)
# Initialize Bedrock client with aws_sdk_bedrock_runtime
config: Final = Config(
credentials_identity: Final = AWSCredentialsIdentity(
access_key_id=frozen_credentials.access_key,
secret_access_key=frozen_credentials.secret_key,
session_token=frozen_credentials.token,
)
config: Final = await AsyncBedrockRuntimeConfig.resolve(
endpoint_uri=endpoint_uri,
region=aws_region_name,
aws_access_key_id=frozen_credentials.access_key,
aws_secret_access_key=frozen_credentials.secret_key,
aws_session_token=frozen_credentials.token,
aws_credentials_identity_resolver=StaticCredentialsResolver(),
aws_credentials_identity_resolver=StaticCredentialsResolver(identity=credentials_identity),
transport=AWSCRTHTTPClient(),
)
bedrock_client: Final = BedrockRuntimeClient(config=config)
bedrock_client: Final = AsyncBedrockRuntimeClient(config=config)
async def open_bidirectional_stream() -> BedrockBidirectionalStream:
return await bedrock_client.invoke_model_with_bidirectional_stream(
InvokeModelWithBidirectionalStreamOperationInput(model_id=model)
)
try:
await self._run_session(websocket, open_bidirectional_stream, model, logging_obj, pending_session_update)
finally:
await _close_bedrock_client(bedrock_client)
async def _run_session(
self,
websocket: RealtimeClientWebSocket,
open_bidirectional_stream: Callable[[], Awaitable[BedrockBidirectionalStream]],
model: str,
logging_obj: LiteLLMLogging,
pending_session_update: str | None,
) -> None:
transformation_config: Final = BedrockRealtimeConfig()
bedrock_stream: Final = await open_bidirectional_stream()

View file

@ -23788,7 +23788,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
"supports_vision": true
},
"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf": {
"input_cost_per_token": 1.2e-06,
@ -24114,7 +24114,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
"supports_vision": true
},
"fireworks_ai/qwen3p7-plus": {
"cache_read_input_token_cost": 8e-08,
@ -69184,6 +69184,27 @@
"supports_reasoning": true,
"supports_vision": true
},
"typesafe/jev-1.13.0": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-latest": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-preview": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"wandb/zai-org/GLM-5.3-Flash": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_token": 1.5e-07,

View file

@ -208,6 +208,7 @@ LAZY_FEATURES: Final[tuple[LazyFeature, ...]] = (
"/nvidia_nim/",
"/openai/",
"/openai_passthrough/",
"/typesafe/",
"/vertex-ai/",
"/vertex_ai/",
"/vllm/",

View file

@ -20373,6 +20373,96 @@
]
}
},
"/typesafe/{endpoint}": {
"get": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__get",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
},
"post": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__post",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
}
},
"/vertex_ai/discovery/{endpoint}": {
"delete": {
"description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`",
@ -33929,6 +34019,20 @@
"title": "Policy Name",
"type": "string"
},
"priority": {
"anyOf": [
{
"maximum": 2147483647.0,
"minimum": -2147483648.0,
"type": "integer"
},
{
"type": "null"
}
],
"description": "Explicit execution order, lower runs first. Prioritised attachments run before those without one.",
"title": "Priority"
},
"scope": {
"anyOf": [
{
@ -34042,6 +34146,18 @@
"title": "Policy Name",
"type": "string"
},
"priority": {
"anyOf": [
{
"type": "integer"
},
{
"type": "null"
}
],
"description": "Explicit execution order, lower runs first. Prioritised attachments run before those without one.",
"title": "Priority"
},
"scope": {
"anyOf": [
{
@ -36062,6 +36178,20 @@
"title": "Policy Name",
"type": "string"
},
"priority": {
"anyOf": [
{
"maximum": 2147483647.0,
"minimum": -2147483648.0,
"type": "integer"
},
{
"type": "null"
}
],
"description": "Explicit execution order, lower runs first. Prioritised attachments run before those without one.",
"title": "Priority"
},
"scope": {
"anyOf": [
{
@ -38680,8 +38810,7 @@
"required": false,
"schema": {
"default": 10,
"maximum": 100,
"minimum": 1,
"minimum": 0,
"title": "Count",
"type": "integer"
}
@ -39385,8 +39514,7 @@
"required": false,
"schema": {
"default": 10,
"maximum": 100,
"minimum": 1,
"minimum": 0,
"title": "Count",
"type": "integer"
}

View file

@ -51,6 +51,7 @@ from litellm.types.router import RouterErrors, UpdateRouterConfig
from litellm.types.router_weights import validate_router_settings_dict
from litellm.types.secret_managers.main import KeyManagementSystem
from litellm.types.utils import (
AzureSpillover,
CallTypes,
CostBreakdown,
EmbeddingResponse,
@ -483,6 +484,7 @@ class LiteLLMRoutes(enum.Enum):
"/eu.assemblyai",
"/vllm",
"/mistral",
"/typesafe",
"/milvus",
"/gigachat",
"/watsonx",
@ -850,6 +852,7 @@ class LiteLLMRoutes(enum.Enum):
"/team/member_add",
"/team/member_delete",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/member_update",
"/team/{team_id}/member/{user_id}/reset_spend",
"/team/permissions_list",
@ -3895,6 +3898,7 @@ class SpendLogsMetadata(TypedDict):
autorouter_savings: ReadOnly[float | None] # stamped by the logging payload; None = not auto-routed
litellm_gateway_injected_cache: ReadOnly[str | None]
router_metadata: ReadOnly[SpendLogsRouterMetadata | None] # None = deployment not flagged internal_router_model
azure_spillover: ReadOnly[AzureSpillover | None] # None = Azure did not report spillover
class SpendLogsPayload(TypedDict):

View file

@ -17,6 +17,7 @@ if TYPE_CHECKING:
AUTO_ROUTER_LICENSE_FEATURE: Final = "auto_router"
LICENSE_ALL_FEATURES: Final = "*"
AUTO_ROUTER_LICENSE_REMEDY: Final = "A LiteLLM license with the 'auto_router' feature lifts the limit."
@ -153,17 +154,21 @@ class LicenseCheck:
return False
return team_count > _max_teams_in_license
def grants_feature(self, feature: str) -> bool:
if self.airgapped_license_data is None:
return False
allowed_features: Final = self.airgapped_license_data.get("allowed_features")
granted: Final = allowed_features if isinstance(allowed_features, list) else (allowed_features,)
return feature in granted or LICENSE_ALL_FEATURES in granted
def auto_router_capability_limit(self) -> int | None:
"""
How many auto-routers may claim each gated classifier or customization capability:
unlimited (None) only when the signed license lists the auto_router
feature, otherwise one per capability. A license verified through the API carries no
feature list, so it does not lift the limit either.
unlimited (None) only when the signed license lists the auto_router feature or the
"*" wildcard that grants every feature, otherwise one per capability. A license verified
through the API carries no feature list, so it does not lift the limit either.
"""
if self.airgapped_license_data is None:
return 1
allowed_features: Final = self.airgapped_license_data.get("allowed_features")
if isinstance(allowed_features, list) and AUTO_ROUTER_LICENSE_FEATURE in allowed_features:
if self.grants_feature(AUTO_ROUTER_LICENSE_FEATURE):
return None
return 1

View file

@ -31,6 +31,7 @@ _PROXY_ADMIN_VIEW_ONLY_BLOCKED_ROUTES: Final = frozenset(
# team
"/team/new",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/update",
"/team/delete",
"/team/block",
@ -767,6 +768,7 @@ class RouteChecks:
"/user/bulk_update",
"/team/new",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/update",
"/team/delete",
"/model/new",

View file

@ -569,15 +569,15 @@ lite --base-url https://your-proxy.example.com configure claude --api-key sk-...
claude
```
The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-<UTF-8 hex of the group name>` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute up` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control
The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-<UTF-8 hex of the group name>` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute start` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control
Plain `lite configure`, with no agent named, asks which agents to wire and which gateway model each starts on, picked from `/v1/models` with a type-to-filter prompt. All choices and selected config files are checked before the first settings write. If a later filesystem write fails, the output identifies each agent already configured and its undo command
What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute up` session holds a backup, and that check comes before any request
What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute start` session holds a backup, and that check comes before any request
#### Routed model and savings in the status line
`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute up` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline:
`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute start` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline:
```
Routed to: claude-haiku-4-5 -63% vs Claude Opus 5
@ -597,7 +597,7 @@ After upgrading the CLI, rerun your original `lite configure claude` command wit
#### Install the CLI
`lite autoroute up` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
`lite autoroute start` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
```bash
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install.sh | sh
@ -610,7 +610,7 @@ curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch-or-commit>/
LITELLM_CLI_REF=<branch-or-commit> sh
```
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute up`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute start`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
Point the CLI at your real proxy and key before running any `lite model-groups` or `lite autoroute` command -- like every other command in this CLI, they read `LITELLM_PROXY_URL`/`LITELLM_PROXY_API_KEY` (or `--base-url`/`--api-key`), no `lite login` required:
@ -637,44 +637,46 @@ An interactive wizard. It runs the same model-group discovery as above, splits t
The wizard writes the result to `~/.litellm/autorouter/config.yaml` with `0600` permissions, since the file embeds your real proxy API key. Every model referenced anywhere in that config -- tier targets, the classifier model, the embedding model -- becomes its own `litellm_proxy/<model-name>` deployment whose `api_base` and `api_key` point back at your real proxy. That is the trick that keeps your real proxy's config untouched: every actual network call this generates, whether it is the routed completion, an LLM-classifier call, or an embedding call, forwards transparently through your real, already-running proxy with your real key.
You do not need to tell Claude Code to request `autorouter` by name yourself: `lite autoroute up` also sets the top-level `model` and `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`, `ANTHROPIC_DEFAULT_OPUS_MODEL` and `ANTHROPIC_DEFAULT_FABLE_MODEL` to `autorouter` in `~/.claude/settings.json` (and `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when missing, like every other wiring), so every one of Claude Code's own model tiers requests it directly regardless of `/model` or whatever it defaults to otherwise. (A bare `model_name: "*"` deployment looks like the obvious way to catch any request instead, but litellm's Router looks up auto-router deployments by the literal requested model string with no wildcard resolution, so a `"*"` entry would never actually match real traffic -- these env var overrides are what makes it work.)
You do not need to tell Claude Code to request `autorouter` by name yourself: `lite autoroute start` also sets the top-level `model` and `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`, `ANTHROPIC_DEFAULT_OPUS_MODEL` and `ANTHROPIC_DEFAULT_FABLE_MODEL` to `autorouter` in `~/.claude/settings.json` (and `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when missing, like every other wiring), so every one of Claude Code's own model tiers requests it directly regardless of `/model` or whatever it defaults to otherwise. (A bare `model_name: "*"` deployment looks like the obvious way to catch any request instead, but litellm's Router looks up auto-router deployments by the literal requested model string with no wildcard resolution, so a `"*"` entry would never actually match real traffic -- these env var overrides are what makes it work.)
You must run `configure` at least once before `up`; running `up` first fails with a clear error telling you to configure first.
You must run `configure` at least once before `start`; running `start` first fails with a clear error telling you to configure first.
#### Launch the Ephemeral Auto-Router Proxy
```bash
lite autoroute up
lite autoroute start
```
Starts a local, throwaway litellm proxy on `127.0.0.1:5483` (override with `--port`), running the config `configure` generated, with a self-issued API key baked in (your real proxy key never leaves the generated config -- it only appears there, forwarding to your real proxy). Both the port and the key are stable across runs: the key is minted once, persisted inside the generated config, and reused by every later `up` (and carried forward when you re-run `configure`), so anything you configured against one session keeps working in the next. If the port is already taken, `up` refuses with a clear error instead of silently moving to another one. It waits for the ephemeral proxy to report healthy, then patches `~/.claude/settings.json` the same way `lite up` does, except with a static `ANTHROPIC_AUTH_TOKEN` env var instead of an `apiKeyHelper`, since this key is self-issued rather than something needing SSO refresh. Any `claude` session started afterward, from any terminal, routes through the ephemeral proxy.
Starts a local, throwaway litellm proxy on `127.0.0.1:5483` (override with `--port`), running the config `configure` generated, with a self-issued API key baked in (your real proxy key never leaves the generated config -- it only appears there, forwarding to your real proxy). Both the port and the key are stable across runs: the key is minted once, persisted inside the generated config, and reused by every later `start` (and carried forward when you re-run `configure`), so anything you configured against one session keeps working in the next. If the port is already taken, `start` refuses with a clear error instead of silently moving to another one. It waits for the ephemeral proxy to report healthy, then patches `~/.claude/settings.json` the same way `lite up` does, except with a static `ANTHROPIC_AUTH_TOKEN` env var instead of an `apiKeyHelper`, since this key is self-issued rather than something needing SSO refresh. Any `claude` session started afterward, from any terminal, routes through the ephemeral proxy.
`lite autoroute up` runs in the foreground and streams the ephemeral proxy's own log file into your terminal, so you can watch its routing decisions -- which tier and model got picked for each request -- as you use Claude Code normally. Press Ctrl-C (or send SIGTERM) to stop it; this kills the child proxy process and restores your original Claude Code settings, in that order.
`lite autoroute start` runs in the foreground and streams the ephemeral proxy's own log file into your terminal, so you can watch its routing decisions -- which tier and model got picked for each request -- as you use Claude Code normally. Press Ctrl-C (or send SIGTERM) to stop it; this kills the child proxy process and restores your original Claude Code settings, in that order.
#### Recover From an Unclean Shutdown
```bash
lite autoroute down
lite autoroute stop
```
If the `lite autoroute up` process dies uncleanly -- `kill -9`, a crash -- rather than being stopped with Ctrl-C, `down` is the manual recovery path: it kills any leftover ephemeral proxy process found via a recorded pid file and restores Claude Code's settings from whatever backup is on disk.
If the `lite autoroute start` process dies uncleanly -- `kill -9`, a crash -- rather than being stopped with Ctrl-C, `stop` is the manual recovery path: it kills any leftover ephemeral proxy process found via a recorded pid file and restores Claude Code's settings from whatever backup is on disk.
#### Example
```bash
lite autoroute configure
lite autoroute up
lite autoroute start
# use Claude Code as normal in another terminal; routing decisions stream live
lite autoroute down # only needed if `up` was killed uncleanly instead of Ctrl-C'd
lite autoroute stop # only needed if `start` was killed uncleanly instead of Ctrl-C'd
```
The previous names, `lite autoroute up` and `lite autoroute down`, still work as hidden aliases of `start` and `stop`: each prints a deprecation notice on stderr and will be removed in a future release
#### Caveats
Adaptive mode's learned state does not persist across `lite autoroute up` sessions -- there is no local database, so every session starts adaptive selection cold. A Claude Code session already running before `up` started, or still running when it stops, keeps whatever settings it loaded at its own startup; like `lite up`, this is a one-time file patch and restore, not a live traffic interceptor. Only Claude Code is supported, for the same reason as `lite up`: no other supported agent (for example Cursor) has an equivalent hot-patchable config file.
Adaptive mode's learned state does not persist across `lite autoroute start` sessions -- there is no local database, so every session starts adaptive selection cold. A Claude Code session already running before `start` ran, or still running when it stops, keeps whatever settings it loaded at its own startup; like `lite up`, this is a one-time file patch and restore, not a live traffic interceptor. Only Claude Code is supported, for the same reason as `lite up`: no other supported agent (for example Cursor) has an equivalent hot-patchable config file.
A session that outlives `up` (or is still running the moment you stop it) keeps sending requests, master key included, to that now-freed loopback port until you restart it. Once the ephemeral proxy process exits, nothing stops another local account on the same machine from binding that same port and receiving those requests instead -- and since the port is a fixed, predictable default and the master key is a static value that persists across sessions (unlike `lite up`'s `apiKeyHelper`, which is re-resolved per request), whoever receives them gets a live-looking token along with the prompt content. Restart any Claude Code session before you consider the machine clean, run `lite autoroute down` promptly rather than leaving a stopped session's settings patched, and do not run `lite autoroute up` on a shared or multi-tenant host. To rotate the persisted key, delete the `master_key` line from `~/.litellm/autorouter/config.yaml`; the next `up` mints a fresh one (deleting the whole file works too, but then `configure` must be re-run first).
A session that outlives `start` (or is still running the moment you stop it) keeps sending requests, master key included, to that now-freed loopback port until you restart it. Once the ephemeral proxy process exits, nothing stops another local account on the same machine from binding that same port and receiving those requests instead -- and since the port is a fixed, predictable default and the master key is a static value that persists across sessions (unlike `lite up`'s `apiKeyHelper`, which is re-resolved per request), whoever receives them gets a live-looking token along with the prompt content. Restart any Claude Code session before you consider the machine clean, run `lite autoroute stop` promptly rather than leaving a stopped session's settings patched, and do not run `lite autoroute start` on a shared or multi-tenant host. To rotate the persisted key, delete the `master_key` line from `~/.litellm/autorouter/config.yaml`; the next `start` mints a fresh one (deleting the whole file works too, but then `configure` must be re-run first).
Do not run `lite up` and `lite autoroute up` at the same time. Each patches `~/.claude/settings.json` and keeps its own separate backup, with no coordination between them: whichever one you stop or crash out of last is the one whose backup gets restored, which can silently leave the *other* mode's settings (a static master key and a now-dead loopback URL, or a stale `apiKeyHelper`) active. Run `lite down` or `lite autoroute down` (whichever applies) before switching to the other mode.
Do not run `lite up` and `lite autoroute start` at the same time. Each patches `~/.claude/settings.json` and keeps its own separate backup, with no coordination between them: whichever one you stop or crash out of last is the one whose backup gets restored, which can silently leave the *other* mode's settings (a static master key and a now-dead loopback URL, or a stale `apiKeyHelper`) active. Run `lite down` or `lite autoroute stop` (whichever applies) before switching to the other mode.
## Environment Variables

View file

@ -1,5 +1,5 @@
"""CLI package for LiteLLM Proxy Client."""
from .main import cli
from .main import cli, litellm_proxy_cli
__all__ = ["cli"]
__all__ = ["cli", "litellm_proxy_cli"]

View file

@ -51,7 +51,7 @@ def _ensure_master_key() -> str:
The generated config is the single home of the key: the proxy server authenticates against
general_settings.master_key only (a key under litellm_settings is silently ignored, which
would leave the ephemeral proxy with no real auth), and the file is written 0600 via
secure_create. Reusing that persisted value keeps the key stable across `up` runs, so a
secure_create. Reusing that persisted value keeps the key stable across `start` runs, so a
client configured against one session keeps working in the next.
"""
with open(CONFIG_PATH, "r") as f:
@ -88,15 +88,18 @@ def configure(ctx: click.Context) -> None:
run_configure_wizard(ctx)
@autoroute_group.command("up")
@click.option(
_PORT_OPTION: Final = click.option(
"--port",
type=click.IntRange(1, 65535),
default=DEFAULT_AUTOROUTE_PORT,
show_default=True,
help="Loopback port for the ephemeral proxy; stable across runs so configured clients keep working.",
)
def up(port: int) -> None:
@autoroute_group.command("start")
@_PORT_OPTION
def start(port: int) -> None:
"""Launch the ephemeral auto-router proxy and route Claude Code through it"""
if not CONFIG_PATH.exists():
raise click.ClickException("No config found. Run `lite autoroute configure` first.")
@ -104,7 +107,7 @@ def up(port: int) -> None:
missing: Final = missing_proxy_runtime_modules()
if missing:
raise click.ClickException(
"lite autoroute up launches a local litellm proxy, which needs the proxy runtime that the "
"lite autoroute start launches a local litellm proxy, which needs the proxy runtime that the "
f"thin `litellm[cli]` install does not include (missing: {', '.join(missing)}). Install the "
"proxy runtime with `uv tool install --force 'litellm[proxy]'`, or to QA a branch, "
"`curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch>/scripts/install.sh | "
@ -117,14 +120,14 @@ def up(port: int) -> None:
raise click.ClickException(str(e))
if existing_pid is not None and is_running(existing_pid.pid):
raise click.ClickException(
"An ephemeral proxy is already running (lite autoroute up looks already active). "
"Run `lite autoroute down` first."
"An ephemeral proxy is already running (lite autoroute start looks already active). "
"Run `lite autoroute stop` first."
)
if AUTOROUTE_BACKUP_PATH.exists():
raise click.ClickException(
f"{AUTOROUTE_BACKUP_PATH} already exists -- `lite autoroute up` looks like it's already "
"running (or crashed without cleanup). Run `lite autoroute down` first."
f"{AUTOROUTE_BACKUP_PATH} already exists -- `lite autoroute start` looks like it's already "
"running (or crashed without cleanup). Run `lite autoroute stop` first."
)
if port == 4000:
@ -135,8 +138,8 @@ def up(port: int) -> None:
if not is_port_available(port):
raise click.ClickException(
f"Port {port} on 127.0.0.1 is already in use. If a previous `lite autoroute up` is still "
"running or crashed, run `lite autoroute down`; otherwise pick a different port with --port."
f"Port {port} on 127.0.0.1 is already in use. If a previous `lite autoroute start` is still "
"running or crashed, run `lite autoroute stop`; otherwise pick a different port with --port."
)
master_key: Final = _ensure_master_key()
@ -196,7 +199,7 @@ def up(port: int) -> None:
click.echo("\nStopped ephemeral proxy and restored Claude Code settings.")
click.echo(
f"Restart any Claude Code session still open from this session, or another local account could "
f"bind the now-free port {port} and receive its requests. Do not use `lite autoroute up` on a "
f"bind the now-free port {port} and receive its requests. Do not use `lite autoroute start` on a "
f"shared or multi-tenant host."
)
@ -214,13 +217,13 @@ def up(port: int) -> None:
_teardown()
@autoroute_group.command("down")
def down() -> None:
@autoroute_group.command("stop")
def stop() -> None:
"""Restore Claude Code settings and stop a leftover ephemeral proxy, if any"""
try:
record: PidRecord | None = read_pid_record()
except ClaudeSettingsError as e:
# down is the crash-recovery path -- a corrupt pid record must not block it; clear the
# stop is the crash-recovery path -- a corrupt pid record must not block it; clear the
# unusable record and keep going rather than leaving the user with no way to clean up.
click.echo(f"{e} Clearing it and continuing cleanup.", err=True)
record = None
@ -238,7 +241,34 @@ def down() -> None:
elif restored.existed:
click.echo(f"Restored {CLAUDE_SETTINGS_PATH} to its original contents.")
else:
click.echo(f"Removed {CLAUDE_SETTINGS_PATH} (it did not exist before `lite autoroute up`).")
click.echo(f"Removed {CLAUDE_SETTINGS_PATH} (it did not exist before `lite autoroute start`).")
AUTOROUTE_ALIAS_DEPRECATION_NOTICE: Final = (
"`lite autoroute {retired}` is deprecated and will be removed in a future release; "
"run `lite autoroute {current}` instead, it takes the same options."
)
def _warn_deprecated_alias(retired: str, current: str) -> None:
click.secho(AUTOROUTE_ALIAS_DEPRECATION_NOTICE.format(retired=retired, current=current), err=True, fg="yellow")
@autoroute_group.command("up", hidden=True)
@_PORT_OPTION
@click.pass_context
def up(ctx: click.Context, port: int) -> None:
"""Deprecated alias of `lite autoroute start`"""
_warn_deprecated_alias("up", "start")
ctx.invoke(start, port=port)
@autoroute_group.command("down", hidden=True)
@click.pass_context
def down(ctx: click.Context) -> None:
"""Deprecated alias of `lite autoroute stop`"""
_warn_deprecated_alias("down", "stop")
ctx.invoke(stop)
__all__ = ["autoroute_group"]

View file

@ -214,7 +214,7 @@ def build_generated_proxy_config(config: AutorouteConfig, master_key: str) -> di
def master_key_from_config(config: dict[str, JsonValue]) -> str | None:
"""The master key persisted in a generated config, or None when absent or blank.
Single definition of "this config already has a usable key", shared by `up` (reuse
Single definition of "this config already has a usable key", shared by `start` (reuse
instead of minting) and the configure wizard (carry the key forward on rewrite) so the
two sites can never disagree on what counts as one. Returned verbatim, never stripped:
the proxy authenticates against the exact bytes under general_settings.master_key, so a

View file

@ -43,12 +43,12 @@ _PROXY_RUNTIME_MODULES: tuple[str, ...] = ("fastapi", "uvicorn", "backoff", "orj
def missing_proxy_runtime_modules() -> tuple[str, ...]:
"""Proxy-server modules that ``lite autoroute up`` needs but the thin CLI install lacks.
"""Proxy-server modules that ``lite autoroute start`` needs but the thin CLI install lacks.
``launch_proxy`` runs the full ``litellm.proxy.proxy_cli`` server, whose dependencies live in
the ``proxy`` extra, not the ``cli`` extra that installs the ``lite`` command. On a thin
``litellm[cli]`` install the subprocess dies with a bare ``ModuleNotFoundError``; detecting the
gap here lets ``up`` fail with an actionable message instead.
gap here lets ``start`` fail with an actionable message instead.
"""
return tuple(name for name in _PROXY_RUNTIME_MODULES if importlib.util.find_spec(name) is None)

View file

@ -94,7 +94,7 @@ def _load_persisted_master_key(config_path: Path) -> str | None:
"""The master key from an existing generated config, so a rewrite carries it forward.
Lenient on a missing or corrupt file: configure is the regeneration path, so it must
succeed from any prior state; a key that cannot be read is simply not carried and `up`
succeed from any prior state; a key that cannot be read is simply not carried and `start`
mints a fresh one.
"""
if not config_path.exists():

View file

@ -1,6 +1,6 @@
"""Shared handling of Claude Code's ~/.claude/settings.json.
`lite up` and `lite autoroute up` patch this file temporarily and restore it on
`lite up` and `lite autoroute start` patch this file temporarily and restore it on
exit; `lite configure claude` patches it persistently and records how to undo it.
All of them need the same merge, and `up` already imports from `auth`, so the
shared parts live here rather than in any one command module. The credential is
@ -88,7 +88,7 @@ class SettingsFileOwner:
SETTINGS_FILE_OWNERS: Final = (
SettingsFileOwner(BACKUP_PATH, "lite up", "lite down"),
SettingsFileOwner(AUTOROUTE_BACKUP_PATH, "lite autoroute up", "lite autoroute down"),
SettingsFileOwner(AUTOROUTE_BACKUP_PATH, "lite autoroute start", "lite autoroute stop"),
)
_SETTINGS_ADAPTER: Final = TypeAdapter(dict[str, JsonValue])
@ -111,7 +111,7 @@ def _is_default_settings_file(settings_path: Path) -> bool:
def settings_file_owners(settings_path: Path) -> tuple[SettingsFileOwner, ...]:
"""The commands whose backups guard settings_path: `lite up` and `lite autoroute up` only ever manage the default file."""
"""The commands whose backups guard settings_path: `lite up` and `lite autoroute start` only ever manage the default file."""
return SETTINGS_FILE_OWNERS if _is_default_settings_file(settings_path) else ()
@ -240,7 +240,7 @@ def _env_object(settings: Mapping[str, JsonValue], path: Path) -> Mapping[str, J
def refuse_while_owned(settings_path: Path, owners: Sequence[SettingsFileOwner]) -> None:
"""Refuse while `lite up` or `lite autoroute up` holds a backup it will restore over any write; a
"""Refuse while `lite up` or `lite autoroute start` holds a backup it will restore over any write; a
purely local check, so commands run it before any login prompt or request."""
for owner in owners:
if owner.backup_path.exists():
@ -262,7 +262,7 @@ def _write_target(settings_path: Path) -> Path:
def write_claude_settings(settings_path: Path, settings: Mapping[str, JsonValue]) -> None:
"""The one way a settings document lands on disk: staged owner-only beside the target and renamed into
place, through a symlink rather than over it. Every writer (`configure`, `up`, `autoroute up` and the
place, through a symlink rather than over it. Every writer (`configure`, `up`, `autoroute start` and the
restores) may be carrying the credential, so none creates the file under the umask or truncates it."""
target: Final = _write_target(settings_path)
try:
@ -341,7 +341,7 @@ def merge_claude_settings(
an apiKeyHelper) are removed, since Claude Code given two credentials may send the wrong one.
ENABLE_TOOL_SEARCH and CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY get their defaults only when
missing. `default_model` is the top-level `model` and env.ANTHROPIC_MODEL (see StartOn);
`tier_model` is `lite autoroute up`'s knob that points every ANTHROPIC_DEFAULT_*_MODEL at one
`tier_model` is `lite autoroute start`'s knob that points every ANTHROPIC_DEFAULT_*_MODEL at one
group. Apart from those tier keys, exactly OWNED_PATHS are touched.
"""
raw_env: Final = settings.get(ENV_KEY, {})

View file

@ -56,7 +56,7 @@ _CLAUDE_CODE_VIEW: Final = MappingProxyType(
_MODEL_OPTION_HELP: Final = (
f"Proxy model to set as {STARTING_MODEL_ROLE}. Must be listed on /v1/models for the key; without it, "
"Claude Code keeps its own default and a pin an earlier configure made is let go of. Nothing pins Claude "
"Code's sub-agent or background tiers; `lite autoroute up` is the mode that does."
"Code's sub-agent or background tiers; `lite autoroute start` is the mode that does."
)

View file

@ -36,8 +36,8 @@ def migrate(ctx: click.Context, check_only: bool, dry_run: bool):
resumable; safe to re-run after an interruption.
Examples:
litellm-proxy encryption migrate --check # attestation scan, no writes
litellm-proxy encryption migrate # perform the migration
lite encryption migrate --check # attestation scan, no writes
lite encryption migrate # perform the migration
"""
client: Final = HTTPClient(ctx.obj["base_url"], ctx.obj["api_key"])

View file

@ -168,5 +168,16 @@ cli.add_command(configure_group)
cli.add_command(unconfigure_group)
LITELLM_PROXY_DEPRECATION_NOTICE: Final = (
"The `litellm-proxy` command is deprecated and will be removed in a future release; "
"run `lite` instead, it takes the same commands and options."
)
def litellm_proxy_cli() -> None:
click.secho(LITELLM_PROXY_DEPRECATION_NOTICE, err=True, fg="yellow")
cli()
if __name__ == "__main__":
cli()

View file

@ -78,3 +78,27 @@ def get_budget_reset_time(budget_duration: str) -> datetime:
`BudgetResetSettings` by injection (creation/update endpoints, startup backfill).
"""
return compute_budget_reset_at(budget_duration, get_budget_reset_settings())
def _is_persistable_budget_duration(budget_duration: str) -> bool:
from litellm.litellm_core_utils.duration_parser import duration_in_seconds
try:
if duration_in_seconds(budget_duration) <= 0:
return False
get_budget_reset_time(budget_duration=budget_duration)
except (ValueError, OverflowError):
return False
return True
def budget_duration_error(budget_duration: str | None) -> str | None:
"""Why `budget_duration` cannot be persisted, or None when it is usable.
A non-positive duration resolves to a reset time of "now", which leaves the row
permanently due: the reset job re-reads it every tick and, once enough of them
exist, they fill each batch and starve every other tenant's reset.
"""
if budget_duration is None or _is_persistable_budget_duration(budget_duration):
return None
return f"Invalid budget_duration '{budget_duration}'. Use a format like '1h', '24h', '7d', or '30d'."

View file

@ -22,7 +22,7 @@ from litellm.types.llms.openai import (
BaseLiteLLMOpenAIResponseObject,
ResponsesAPIResponse,
)
from litellm.types.utils import CallTypesLiteral, LLMResponseTypes, SpecialEnums
from litellm.types.utils import ADDRESSED_RESPONSE_ID_FIELD, CallTypesLiteral, LLMResponseTypes, SpecialEnums
if TYPE_CHECKING:
from litellm.caching.caching import DualCache
@ -32,7 +32,6 @@ if TYPE_CHECKING:
_RESPONSES_API_PROVIDER_PREFIX: Final = "/openai"
_RESPONSES_API_CREATE_ROUTES: Final = frozenset({"/v1/responses", "/responses"})
_ADDRESSED_RESPONSE_ID_KEY: Final = "_litellm_addressed_response_id"
_UNMANAGED_RESPONSE_ID_DETAIL: Final = (
"Forbidden. This response id was not issued by this proxy, so the proxy cannot tell who owns it. "
"To let keys address responses this proxy did not issue, set "
@ -132,7 +131,7 @@ class ResponsesIDSecurity(CustomLogger):
if call_type not in responses_api_call_types:
return None
addressed_id_field: Final = "previous_response_id" if call_type == "aresponses" else "response_id"
retained_id: Final = data.get(_ADDRESSED_RESPONSE_ID_KEY)
retained_id: Final = data.get(ADDRESSED_RESPONSE_ID_FIELD)
addressed_id: Final = (
retained_id if isinstance(retained_id, str) and retained_id else data.get(addressed_id_field)
)
@ -140,7 +139,7 @@ class ResponsesIDSecurity(CustomLogger):
return data
authorized_id: Final = self._authorize_response_id(addressed_id, user_api_key_dict)
data[addressed_id_field] = authorized_id
data[_ADDRESSED_RESPONSE_ID_KEY] = addressed_id
data[ADDRESSED_RESPONSE_ID_FIELD] = addressed_id
return data
def _authorize_response_id(

View file

@ -36,6 +36,10 @@ router: Final = APIRouter()
IMAGE_EDIT_NUMERIC_FORM_FIELDS: Final = numeric_form_fields(get_type_hints(ImageEditRequestParams))
IMAGE_ARRAY_FIELD: Final = "image[]"
MASK_ARRAY_FIELD: Final = "mask[]"
BRACKETED_FILE_FIELDS: Final = frozenset({IMAGE_ARRAY_FIELD, MASK_ARRAY_FIELD})
async def uploadfile_to_bytesio(upload: UploadFile) -> io.BytesIO:
"""
@ -244,9 +248,9 @@ async def image_edit_api(
fastapi_response: Response,
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
image: list[UploadFile] | None = File(None),
image_array: list[UploadFile] | None = File(None, alias="image[]"),
image_array: list[UploadFile] | None = File(None, alias=IMAGE_ARRAY_FIELD),
mask: list[UploadFile] | None = File(None),
mask_array: list[UploadFile] | None = File(None, alias="mask[]"),
mask_array: list[UploadFile] | None = File(None, alias=MASK_ARRAY_FIELD),
model: str | None = None,
):
"""
@ -294,12 +298,14 @@ async def image_edit_api(
#########################################################
# Read request body and convert UploadFiles to BytesIO
#########################################################
data: Final = dict(
coerce_numeric_form_fields(
data: Final = {
key: value
for key, value in coerce_numeric_form_fields(
parsed_body=await _read_request_body(request=request),
numeric_fields=IMAGE_EDIT_NUMERIC_FORM_FIELDS,
)
)
).items()
if key not in BRACKETED_FILE_FIELDS
}
image_files: Final = await batch_to_bytesio(image)
mask_files: Final = await batch_to_bytesio(mask)
if image_files:

View file

@ -1,5 +1,6 @@
import math
from collections.abc import Mapping
from types import MappingProxyType
from typing import TYPE_CHECKING, Any, Final, Optional, Union
from fastapi import HTTPException, status
@ -33,23 +34,11 @@ def validate_budget_duration(budget_duration: str | None, status_code: int = 400
enough of them exist, they fill each batch and starve every other tenant's
reset.
"""
if budget_duration is None:
return
from litellm.proxy.common_utils.timezone_utils import budget_duration_error
from litellm.litellm_core_utils.duration_parser import duration_in_seconds
from litellm.proxy.common_utils.timezone_utils import get_budget_reset_time
try:
if duration_in_seconds(budget_duration) <= 0:
raise ValueError("budget_duration must be positive")
get_budget_reset_time(budget_duration=budget_duration)
except (ValueError, OverflowError):
raise HTTPException(
status_code=status_code,
detail={
"error": f"Invalid budget_duration '{budget_duration}'. Use a format like '1h', '24h', '7d', or '30d'."
},
)
error: Final = budget_duration_error(budget_duration)
if error is not None:
raise HTTPException(status_code=status_code, detail={"error": error})
from litellm._logging import verbose_proxy_logger
@ -490,6 +479,33 @@ _TEAM_MEMBER_BUDGET_LIMIT_FIELDS: Final = (
)
MEMBER_BUDGET_PATCH_FIELDS: Final = MappingProxyType(
{
"max_budget_in_team": "max_budget",
"tpm_limit": "tpm_limit",
"rpm_limit": "rpm_limit",
"budget_duration": "budget_duration",
"allowed_models": "allowed_models",
}
)
def _prisma_value(value: object) -> object:
return list(value) if isinstance(value, tuple) else value
def member_budget_patch(source: BaseModel) -> dict[str, Any]:
"""Map the per-member limit fields a request actually set to their budget-table
columns (merge-patch: a sent value updates, an explicit null clears, an absent
field is left untouched)."""
provided: Final = source.model_dump(exclude_unset=True)
return {
column: _prisma_value(provided[request_field])
for request_field, column in MEMBER_BUDGET_PATCH_FIELDS.items()
if request_field in provided
}
def _is_set_budget_value(value: object) -> bool:
if value is None:
return False
@ -513,6 +529,7 @@ async def _upsert_budget_and_membership(
user_api_key_dict: UserAPIKeyAuth,
budget_patch: dict[str, Any],
team_default_budget_id: str | None = None,
shared_budget_ids: frozenset[str] | None = None,
):
"""
Apply a merge-patch of per-member budget fields to a team membership.
@ -527,6 +544,10 @@ async def _upsert_budget_and_membership(
(from team metadata.team_member_budget_id). When the membership still
points at it, we clone-on-write so editing one member's budget does not
mutate the shared default that every other member points at.
``shared_budget_ids`` extends that protection to any other row more than one
membership points at, which a caller patching several members at once has
already counted; a row listed there is cloned rather than written in place.
"""
if not budget_patch:
return
@ -538,10 +559,8 @@ async def _upsert_budget_and_membership(
get_budget_reset_time(budget_duration=duration) if duration is not None else None
)
is_shared_default: Final = (
existing_budget_id is not None
and team_default_budget_id is not None
and existing_budget_id == team_default_budget_id
is_shared_default: Final = existing_budget_id is not None and (
existing_budget_id == team_default_budget_id or existing_budget_id in (shared_budget_ids or frozenset())
)
async def _disconnect():
@ -563,25 +582,25 @@ async def _upsert_budget_and_membership(
)
return
create_data: Final[dict[str, Any]] = {
source_row: Final = (
await tx.litellm_budgettable.find_unique(where={"budget_id": existing_budget_id}) if is_shared_default else None
)
source: Final[Mapping[str, Any]] = source_row.model_dump() if source_row is not None else MappingProxyType({})
create_data: Final[dict[str, Any]] = { # mutable-ok: Prisma create payloads are dict-shaped
"created_by": user_api_key_dict.user_id or "",
"updated_by": user_api_key_dict.user_id or "",
**MappingProxyType(
{f: source[f] for f in _TEAM_MEMBER_BUDGET_LIMIT_FIELDS if _is_set_budget_value(source.get(f))}
),
**write_data,
}
if is_shared_default:
default_budget_row: Final = await tx.litellm_budgettable.find_unique(where={"budget_id": existing_budget_id})
if default_budget_row is not None:
default_budget_dict: Final = default_budget_row.model_dump()
for field in _TEAM_MEMBER_BUDGET_LIMIT_FIELDS:
value = default_budget_dict.get(field)
if _is_set_budget_value(value):
create_data[field] = value
create_data.update(write_data)
if create_data.get("budget_duration") is not None:
create_data["budget_reset_at"] = get_budget_reset_time(budget_duration=create_data["budget_duration"])
else:
# Restarting an inherited window on an unrelated edit hands the member a free period.
carried: Final = source.get("budget_reset_at") if "budget_duration" not in budget_patch else None
if carried is not None:
create_data["budget_reset_at"] = carried
if create_data.get("budget_reset_at") is None:
create_data.pop("budget_reset_at", None)
if not _has_meaningful_budget_limit(create_data):

Some files were not shown because too many files have changed in this diff Show more