Merge remote-tracking branch 'origin/main' into litellm_mcp_admin_terminate_sessions_revoke_credentials

This commit is contained in:
yassin 2026-09-18 02:10:58 +00:00
commit e6349d00f3
262 changed files with 12344 additions and 8850 deletions

View file

@ -1,9 +1,9 @@
"""Auto-merge the provider-info-sync bot's cost-map pull requests.
Evaluates every gate (author allowlist, cost-map-only diff, required and
non-required checks, Greptile confidence, Bugbot review, human reviews) and
merges with a merge commit when all of them hold. Every hold reason is
logged; the process exits 0 on hold and 1 only on API or programming errors.
non-required checks, human reviews) and merges with a merge commit when
all of them hold. Every hold reason is logged; the process exits 0 on hold
and 1 only on API or programming errors.
``DRY_RUN=1`` prints the verdict without calling the merge endpoint.
"""
@ -11,7 +11,6 @@ from __future__ import annotations
import json
import os
import re
import subprocess
import sys
import time
@ -27,12 +26,6 @@ CLASSIFY_SCRIPT: Final = os.path.join(REPO_ROOT, ".circleci", "scripts", "classi
API_ROOT: Final = "https://api.github.com"
CHANGED_FILE_CEILING: Final = 3000
OK_CHECK_CONCLUSIONS: Final = frozenset({"success", "skipped", "neutral"})
GREPTILE_LOGIN: Final = "greptile-apps[bot]"
BUGBOT_LOGIN: Final = "cursor[bot]"
GREPTILE_SCORE_RE: Final = re.compile(r"Confidence Score:\s*(\d)/5")
BUGBOT_REVIEW_MARKER: Final = "<!-- BUGBOT_REVIEW -->"
BUGBOT_STALE_MARKER: Final = "<!-- BUGBOT_REVIEW_STALE -->"
BUGBOT_CLEAN: Final = "found no new issues"
@dataclass(frozen=True, slots=True)
@ -60,13 +53,6 @@ class CommitStatus:
state: str
@dataclass(frozen=True, slots=True)
class IssueComment:
author_login: str
body: str
updated_at: datetime
@dataclass(frozen=True, slots=True)
class Review:
author_login: str
@ -89,9 +75,7 @@ class EvaluationInputs:
required_contexts: frozenset[str]
check_runs: tuple[CheckRun, ...]
statuses: tuple[CommitStatus, ...]
comments: tuple[IssueComment, ...]
reviews: tuple[Review, ...]
head_commit_date: datetime
self_check_name: str
author_allowlist: frozenset[str]
@ -155,37 +139,6 @@ def evaluate(
if status.state != "success":
reasons.append(f"commit status {status.context!r} is {status.state}")
greptile: Final = tuple(
comment
for comment in inputs.comments
if comment.author_login == GREPTILE_LOGIN and GREPTILE_SCORE_RE.search(comment.body)
)
if not greptile:
reasons.append("greptile score not available")
else:
latest: Final = max(greptile, key=lambda comment: comment.updated_at)
match: Final = GREPTILE_SCORE_RE.search(latest.body)
score: Final = int(match.group(1)) if match else 0
if latest.updated_at < inputs.head_commit_date:
reasons.append("greptile score older than head commit")
elif score != 5:
reasons.append(f"greptile score {score}/5 below 5")
bugbot: Final = tuple(
review
for review in inputs.reviews
if review.author_login == BUGBOT_LOGIN
and BUGBOT_REVIEW_MARKER in review.body
and BUGBOT_STALE_MARKER not in review.body
and review.commit_id == pr.head_sha
)
if not bugbot:
reasons.append("bugbot review not available")
else:
latest_review: Final = max(bugbot, key=lambda review: review.submitted_at)
if BUGBOT_CLEAN not in latest_review.body:
reasons.append("bugbot reported issues")
latest_state_by_reviewer: Final[dict[str, str]] = {}
for review in sorted(inputs.reviews, key=lambda review: review.submitted_at):
if _is_bot_login(review.author_login):
@ -350,19 +303,6 @@ def _statuses(token: str, repo: str, sha: str) -> tuple[CommitStatus, ...]:
)
def _comments(token: str, repo: str, number: int) -> tuple[IssueComment, ...]:
comments: Final = _paginate(token, f"/repos/{repo}/issues/{number}/comments")
return tuple(
IssueComment(
author_login=_text(_nested(item, "user", "login")),
body=_text(item.get("body")),
updated_at=_parse_time(item.get("updated_at")),
)
for item in comments
if isinstance(item, Mapping)
)
def _reviews(token: str, repo: str, number: int) -> tuple[Review, ...]:
reviews: Final = _paginate(token, f"/repos/{repo}/pulls/{number}/reviews")
return tuple(
@ -378,16 +318,6 @@ def _reviews(token: str, repo: str, number: int) -> tuple[Review, ...]:
)
def _head_commit_date(token: str, repo: str, number: int) -> datetime:
commits: Final = _paginate(token, f"/repos/{repo}/pulls/{number}/commits")
if not commits:
return datetime.min.replace(tzinfo=timezone.utc)
last: Final = commits[-1]
if not isinstance(last, Mapping):
return datetime.min.replace(tzinfo=timezone.utc)
return _parse_time(_nested(last, "commit", "committer", "date"))
def _mergeable_or_refetch(token: str, repo: str, pr: PullRequest) -> PullRequest:
if pr.mergeable is not None:
return pr
@ -410,9 +340,7 @@ def _gather_inputs(
required_contexts=_required_contexts(token, repo, base),
check_runs=_check_runs(token, repo, pr.head_sha),
statuses=_statuses(token, repo, pr.head_sha),
comments=_comments(token, repo, number),
reviews=_reviews(token, repo, number),
head_commit_date=_head_commit_date(token, repo, number),
self_check_name=self_check_name,
author_allowlist=allowlist,
)

View file

@ -94,7 +94,6 @@ jobs:
tests/proxy_unit_tests/test_jwt_key_mapping.py
tests/proxy_unit_tests/test_proxy_custom_auth.py
tests/proxy_unit_tests/test_key_generate_dynamodb.py
tests/proxy_unit_tests/test_deployed_proxy_keygen.py
workers: 4
dist: loadscope
timeout: 15
@ -110,8 +109,6 @@ jobs:
- test-group: proxy-server-core
test-path: >-
tests/proxy_unit_tests/test_proxy_server.py
tests/proxy_unit_tests/test_proxy_server_keys.py
tests/proxy_unit_tests/test_proxy_server_spend.py
tests/proxy_unit_tests/test_aproxy_startup.py
workers: 4
dist: loadscope
@ -120,7 +117,6 @@ jobs:
test-path: >-
tests/proxy_unit_tests/test_proxy_config_unit_test.py
tests/proxy_unit_tests/test_proxy_routes.py
tests/proxy_unit_tests/test_proxy_gunicorn.py
tests/proxy_unit_tests/test_server_root_path.py
tests/proxy_unit_tests/test_proxy_pass_user_config.py
tests/proxy_unit_tests/test_proxy_token_counter.py
@ -198,7 +194,6 @@ jobs:
tests/proxy_unit_tests/test_realtime_cache.py
tests/proxy_unit_tests/test_proxy_exception_mapping.py
tests/proxy_unit_tests/test_custom_tokenizer_bug.py
tests/proxy_unit_tests/test_model_response_typing
workers: 4
dist: loadscope
timeout: 15

View file

@ -1,614 +0,0 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "grafana",
"uid": "-- Grafana --"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"description": "",
"editable": true,
"fiscalYearStartMonth": 0,
"graphTooltip": 0,
"id": 2039,
"links": [],
"liveNow": false,
"panels": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "s"
},
"overrides": []
},
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 0
},
"id": 10,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "histogram_quantile(0.99, sum(rate(litellm_self_latency_bucket{self=\"self\"}[1m])) by (le))",
"legendFormat": "Time to first token",
"range": true,
"refId": "A"
}
],
"title": "Time to first token (latency)",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "currencyUSD"
},
"overrides": [
{
"matcher": {
"id": "byName",
"options": "7e4b0627fd32efdd2313c846325575808aadcf2839f0fde90723aab9ab73c78f"
},
"properties": [
{
"id": "displayName",
"value": "Translata"
}
]
}
]
},
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 8
},
"id": 11,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_spend_metric_total[30d])) by (hashed_api_key)",
"legendFormat": "{{team}}",
"range": true,
"refId": "A"
}
],
"title": "Spend by team",
"transformations": [],
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 9,
"w": 12,
"x": 0,
"y": 16
},
"id": 2,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum by (model) (increase(litellm_requests_metric_total[5m]))",
"legendFormat": "{{model}}",
"range": true,
"refId": "A"
}
],
"title": "Requests by model",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [],
"noValue": "0",
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 3,
"x": 0,
"y": 25
},
"id": 8,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"textMode": "auto"
},
"pluginVersion": "9.4.17",
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_llm_api_failed_requests_metric_total[1h]))",
"legendFormat": "__auto",
"range": true,
"refId": "A"
}
],
"title": "Faild Requests",
"type": "stat"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
},
"unit": "currencyUSD"
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 3,
"x": 3,
"y": 25
},
"id": 6,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_spend_metric_total[30d])) by (model)",
"legendFormat": "{{model}}",
"range": true,
"refId": "A"
}
],
"title": "Spend",
"type": "timeseries"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisCenteredZero": false,
"axisColorMode": "text",
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 6,
"x": 6,
"y": 25
},
"id": 4,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom",
"showLegend": true
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(increase(litellm_total_tokens_total[5m])) by (model)",
"legendFormat": "__auto",
"range": true,
"refId": "A"
}
],
"title": "Tokens",
"type": "timeseries"
}
],
"refresh": "1m",
"revision": 1,
"schemaVersion": 38,
"style": "dark",
"tags": [],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "prometheus",
"value": "edx8memhpd9tsa"
},
"hide": 0,
"includeAll": false,
"label": "datasource",
"multi": false,
"name": "DS_PROMETHEUS",
"options": [],
"query": "prometheus",
"queryValue": "",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
}
]
},
"time": {
"from": "now-1h",
"to": "now"
},
"timepicker": {},
"timezone": "",
"title": "LLM Proxy",
"uid": "rgRrHxESz",
"version": 15,
"weekStart": ""
}

View file

@ -1,6 +0,0 @@
## This folder contains the `json` for creating the following Grafana Dashboard
### Pre-Requisites
- Setup LiteLLM Proxy Prometheus Metrics https://docs.litellm.ai/docs/proxy/prometheus
![1716623265684](https://github.com/BerriAI/litellm/assets/29436595/0e12c57e-4a2d-4850-bd4f-e4294f87a814)

View file

@ -0,0 +1,11 @@
# LiteLLM All Prometheus Metrics dashboard
Every `litellm_*` metric family the proxy can expose on `/metrics` (134 families across 95 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
Import `grafana_dashboard.json` from **Dashboards > New > Import** and pick your Prometheus data source when prompted (the `DS_PROMETHEUS` variable). Counters are plotted as `rate()` over `$__rate_interval`, histograms as p50 / p95 / p99, gauges as the raw value grouped by the most useful label. Every query names the metric exactly as the proxy emits it (counters carry the `_total` suffix the Prometheus client adds), and `tests/test_litellm/integrations/test_prometheus_metric_name_consistency.py` fails if a metric is renamed without updating this dashboard
The first eleven rows need only `callbacks: ["prometheus"]`. The last three rows and the `litellm_admission_*` panels are emitted by other subsystems and stay empty until those are on: the service callback row needs `service_callback: ["prometheus_system"]` in `litellm_settings`, the circuit breaker row needs a Redis cache, the cleanup row needs spend log retention, and admission control needs its middleware enabled. Within the base rows, many panels only fill in once the matching feature is in use: budgets need keys, teams, users or orgs with `max_budget` set, cache panels need caching on, guardrail and MCP panels need those features configured, deployment health needs the router with more than one deployment or a failure to record, and `litellm_in_flight_requests` needs traffic at scrape time. An empty panel for a feature you do not use is expected
## Pre-requisites
Prometheus metrics on the proxy: https://docs.litellm.ai/docs/proxy/prometheus

View file

@ -476,7 +476,7 @@
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "topk(5, sort(litellm_remaining_requests))",
"expr": "topk(5, sort(litellm_remaining_requests_metric))",
"legendFormat": "__auto",
"range": true,
"refId": "A"
@ -573,7 +573,7 @@
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "topk(5, sort(litellm_remaining_tokens))",
"expr": "topk(5, sort(litellm_remaining_tokens_metric))",
"legendFormat": "__auto",
"range": true,
"refId": "A"

View file

@ -6,8 +6,14 @@ This folder contains the `json` for creating Grafana Dashboards
Charts the `gen_ai.*` metrics from the OpenTelemetry v2 integration: spend, tokens, request rate, and latency percentiles by model. Separate from the dashboards below, which chart the `litellm_*` Prometheus metrics.
## [LiteLLM All Prometheus Metrics dashboard](./dashboard_all_metrics)
Every `litellm_*` Prometheus metric family the proxy can emit (134 families, 95 panels) grouped by theme: traffic, latency, spend and tokens, cache, deployments, rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, plus the Redis circuit breaker, spend log cleanup and `prometheus_system` service metrics. Start here if you want everything on one screen; see its [readme](./dashboard_all_metrics/readme.md) for import steps and which panels need a feature enabled before they show data
## [LiteLLM v2 Dashboard](./dashboard_v2)
A compact view of proxy request rate, failures, latency and the top remaining-request / remaining-token gauges per model group
<img width="1316" alt="grafana_1" src="https://github.com/user-attachments/assets/d0df802d-0cb9-4906-a679-941c547789ab">
<img width="1289" alt="grafana_2" src="https://github.com/user-attachments/assets/b11f755f-e113-42ab-b21d-83f91f451a28">
<img width="1323" alt="grafana_3" src="https://github.com/user-attachments/assets/cb29ffdb-477d-4be1-a5cd-c3f7f2cb21c5">

View file

@ -96,6 +96,7 @@ GATEWAY_PATH_PREFIXES: tuple[str, ...] = (
"/langfuse/",
"/vllm/",
"/mistral/",
"/typesafe/",
"/nvidia_nim/",
"/groq/",
"/voyage/",

View file

@ -0,0 +1 @@
ALTER TABLE "LiteLLM_PolicyAttachmentTable" ADD COLUMN IF NOT EXISTS "priority" INTEGER;

View file

@ -1379,6 +1379,7 @@ model LiteLLM_PolicyAttachmentTable {
keys String[] @default([]) // Key aliases or patterns
models String[] @default([]) // Model names or patterns
tags String[] @default([]) // Tag patterns (e.g., ["healthcare", "prod-*"])
priority Int? // Explicit execution order
created_at DateTime @default(now())
created_by String?
updated_at DateTime @default(now()) @updatedAt

View file

@ -2016,6 +2016,7 @@ dependencies = [
"litellm-auth-azure",
"litellm-auth-gcp",
"litellm-framing",
"litellm-providers",
"mime_guess",
"moka",
"rand 0.8.7",
@ -2052,6 +2053,18 @@ dependencies = [
"tokio",
]
[[package]]
name = "litellm-providers"
version = "0.1.0"
dependencies = [
"litellm-auth",
"litellm-auth-aws",
"rstest",
"serde",
"serde_json",
"thiserror 2.0.19",
]
[[package]]
name = "litellm-python-bridge"
version = "0.1.0"

View file

@ -16,6 +16,7 @@ litellm-auth = { path = "crates/auth" }
litellm-auth-aws = { path = "crates/auth-aws" }
litellm-auth-azure = { path = "crates/auth-azure" }
litellm-auth-gcp = { path = "crates/auth-gcp" }
litellm-providers = { path = "crates/providers" }
litellm-cache = { path = "crates/cache" }
litellm-cache-memory = { path = "crates/cache-memory" }
litellm-token-counter = { path = "crates/token-counter" }

View file

@ -15,6 +15,7 @@ litellm-auth.workspace = true
litellm-auth-aws.workspace = true
litellm-auth-azure.workspace = true
litellm-auth-gcp.workspace = true
litellm-providers.workspace = true
litellm-framing.workspace = true
moka.workspace = true
mime_guess = "2.0.5"

View file

@ -24,3 +24,23 @@ pub enum Error {
#[error(transparent)]
Aws(#[from] litellm_auth_aws::Error),
}
impl From<litellm_providers::audio_transcription::Error> for Error {
fn from(error: litellm_providers::audio_transcription::Error) -> Self {
match error {
litellm_providers::audio_transcription::Error::InvalidType { expected, actual } => {
Self::InvalidType { expected, actual }
}
litellm_providers::audio_transcription::Error::MissingField(field) => {
Self::MissingField(field)
}
litellm_providers::audio_transcription::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::audio_transcription::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::audio_transcription::Error::Auth(error) => Self::Auth(error),
}
}
}

View file

@ -47,8 +47,8 @@ async fn signed_headers(
use std::collections::BTreeMap;
use std::time::SystemTime;
use crate::llms::base_llm::audio_transcription::transformation::AudioTranscriptionAuth;
use litellm_auth_aws::{aws_auth_config, resolve_credentials, sign_bedrock_post};
use litellm_providers::base_llm::audio_transcription::transformation::AudioTranscriptionAuth;
let AudioTranscriptionAuth::AwsSigV4 { region, .. } = &request.auth else {
return Ok(request.upstream_headers.clone());

View file

@ -3,7 +3,7 @@ pub use error::Error;
mod client;
mod handler;
mod prepare;
pub mod types;
pub use litellm_providers::audio_transcription::types;
pub use handler::execute_audio_transcription_provider_call;
pub use prepare::prepare_audio_transcription_provider_call;

View file

@ -4,10 +4,10 @@ use crate::http_utils::{has_header, string_headers};
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::audio_transcription::transformation::{
use litellm_providers::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
use crate::llms::bedrock::audio_transcription::BEDROCK_AUDIO_TRANSCRIPTION_CONFIG;
use litellm_providers::bedrock::audio_transcription::BEDROCK_AUDIO_TRANSCRIPTION_CONFIG;
fn provider_config(provider: &str) -> Option<&'static dyn BaseAudioTranscriptionConfig> {
if provider == "bedrock" {

View file

@ -2,8 +2,8 @@ use serde_json::{Map, Value};
use super::Error;
use crate::http_utils::string_headers as shared_string_headers;
use crate::llms::anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG;
use crate::llms::base_llm::chat::transformation::BaseConfig;
use litellm_providers::anthropic::chat::transformation::ANTHROPIC_CHAT_COMPLETIONS_CONFIG;
use litellm_providers::base_llm::chat::transformation::BaseConfig;
const HEADER_CONTEXT: &str = "chat completions";
@ -11,7 +11,7 @@ pub(super) fn chat_completions_provider_config(provider: &str) -> Option<&'stati
match provider {
"anthropic" => Some(&ANTHROPIC_CHAT_COMPLETIONS_CONFIG),
"bedrock" => Some(
&crate::llms::bedrock::chat::converse_transformation::BEDROCK_CHAT_COMPLETIONS_CONFIG,
&litellm_providers::bedrock::chat::converse_transformation::BEDROCK_CHAT_COMPLETIONS_CONFIG,
),
_ => None,
}

View file

@ -24,3 +24,19 @@ pub enum Error {
#[error(transparent)]
Aws(#[from] litellm_auth_aws::Error),
}
impl From<litellm_providers::chat::Error> for Error {
fn from(error: litellm_providers::chat::Error) -> Self {
match error {
litellm_providers::chat::Error::MissingField(field) => Self::MissingField(field),
litellm_providers::chat::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::chat::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::chat::Error::Unsupported(reason) => Self::Unsupported(reason),
litellm_providers::chat::Error::Auth(error) => Self::Auth(error),
}
}
}

View file

@ -8,7 +8,7 @@ use super::types::{
ResolvedChatCompletionsRequest,
};
use crate::http_utils::{http_request, truncate_error_body};
use crate::llms::base_llm::chat::transformation::ChatCompletionsAuth;
use litellm_providers::base_llm::chat::transformation::ChatCompletionsAuth;
pub(super) async fn execute_chat_completions_provider_call(
request: ResolvedChatCompletionsRequest<'_>,
@ -59,6 +59,7 @@ pub(super) async fn execute_chat_completions_provider_call(
request
.config
.transform_response(&request.model, ProviderChatResponseData { body })
.map_err(Error::from)
.map_err(as_response_error)
}

View file

@ -10,12 +10,11 @@ mod error;
pub use error::Error;
mod client;
mod common_utils;
pub mod conversation;
pub use litellm_providers::chat::{conversation, response_utils};
pub(crate) mod handler;
mod prepare;
pub mod response_utils;
pub mod streaming;
pub mod types;
pub use litellm_providers::chat::types;
use handler::execute_chat_completions_provider_call;
use prepare::{parse_messages, resolve_provider_config, resolve_request};

View file

@ -10,7 +10,7 @@ use crate::http_utils::has_header;
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
use litellm_providers::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
pub(super) fn resolve_provider_config<'a>(
model: &'a str,

View file

@ -3,7 +3,7 @@ use serde_json::{Map, Value, json};
use super::Error;
use super::prepare::{prepare_provider_request, resolve_request};
use super::types::{ChatCompletionsRequest, ProviderChatCompletionsRequest};
use crate::llms::base_llm::chat::transformation::ChatCompletionsAuth;
use litellm_providers::base_llm::chat::transformation::ChatCompletionsAuth;
fn prepare_chat_completions_call(
request: ChatCompletionsRequest<'_>,

View file

@ -1,36 +1,4 @@
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct CustomLlmProvider<'a> {
pub model: &'a str,
pub custom_llm_provider: &'a str,
}
pub fn get_custom_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
) -> Option<CustomLlmProvider<'a>> {
if let Some(custom_llm_provider) = custom_llm_provider.filter(|provider| !provider.is_empty()) {
return Some(CustomLlmProvider {
model: strip_custom_llm_provider_prefix(model, custom_llm_provider),
custom_llm_provider,
});
}
let (custom_llm_provider, model) = model.split_once('/')?;
if custom_llm_provider.is_empty() || model.is_empty() {
return None;
}
Some(CustomLlmProvider {
model,
custom_llm_provider,
})
}
fn strip_custom_llm_provider_prefix<'a>(model: &'a str, custom_llm_provider: &str) -> &'a str {
model
.strip_prefix(custom_llm_provider)
.and_then(|model| model.strip_prefix('/'))
.unwrap_or(model)
}
pub use litellm_providers::provider_resolution::{CustomLlmProvider, get_custom_llm_provider};
#[cfg(test)]
mod tests {

View file

@ -1,2 +1 @@
pub mod streaming;
pub mod transformation;

View file

@ -2,16 +2,16 @@ use std::collections::HashMap;
use serde_json::Value;
use super::super::experimental_pass_through::messages::streaming::{
AnthropicContentBlock, AnthropicContentBlockDelta, AnthropicMessagesStreamEvent,
AnthropicStreamUsage,
};
use crate::chat_completions::Error;
use crate::chat_completions::streaming::StreamTransformer;
use crate::chat_completions::types::{
ChatCompletionChunk, ChatCompletionThinkingBlock, ChatCompletionToolCallChunk,
ChatCompletionsUsage,
};
use crate::llms::anthropic::experimental_pass_through::messages::streaming::{
AnthropicContentBlock, AnthropicContentBlockDelta, AnthropicMessagesStreamEvent,
AnthropicStreamUsage,
};
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum AnthropicJsonChunkType {

View file

@ -3,9 +3,9 @@ use serde_json::Value;
use time::OffsetDateTime;
use url::Url;
use crate::llms::anthropic::experimental_pass_through::messages::transformation::resolve_anthropic_api_base;
use crate::messages::Error;
use crate::messages::types::AnthropicMessagesResponse;
use litellm_providers::anthropic::experimental_pass_through::messages::transformation::resolve_anthropic_api_base;
const BATCHES_PATH_SUFFIX: &str = "/v1/messages/batches";

View file

@ -1,4 +1,3 @@
pub mod batches;
pub mod count_tokens;
pub mod streaming;
pub mod transformation;

View file

@ -1,2 +1 @@
pub mod anthropic;
pub(crate) mod ocr;

View file

@ -402,4 +402,54 @@ mod tests {
let error = perform_ocr(request).await.unwrap_err();
assert!(error.to_string().contains("data URI"));
}
struct EchoCallerDocument(Value);
impl OcrHooks for EchoCallerDocument {
fn intercepts_requests(&self) -> bool {
true
}
fn during_call(
&self,
mut request: OcrDuringCallRequest,
) -> OcrHookFuture<'_, OcrDuringCallRequest> {
let document = self.0.clone();
Box::pin(async move {
request.body["document"] = document;
Ok(request)
})
}
}
#[tokio::test]
async fn remote_document_stays_inlined_when_hook_echoes_caller_document() {
let (base, seen, server) = mock_server(vec![
MockResponse::json(json!("served document")),
MockResponse::json(json!({"pages":[{"index":0,"markdown":"hello"}],"usage_info":{"pages_processed":1}})),
])
.await;
let document_url = format!("{base}/document.pdf");
let mut request = crate::ocr::test_support::with_source(
wire_request("azure_ai/model", &base, json!({})),
&document_url,
);
request.hooks = Arc::new(EchoCallerDocument(
json!({"type":"document_url","document_url":document_url}),
));
let result = perform_ocr(request).await.unwrap();
server.await.unwrap();
assert_eq!(result.pages[0].markdown, "hello");
let requests = seen.lock().unwrap();
assert_eq!(requests.len(), 2);
assert!(requests[0].starts_with("GET /document.pdf "));
let body: Value =
serde_json::from_str(requests[1].split_once("\r\n\r\n").unwrap().1).unwrap();
assert_eq!(
body["document"]["document_url"],
json!("data:application/json;base64,InNlcnZlZCBkb2N1bWVudCI=")
);
}
}

View file

@ -1,4 +1 @@
pub mod anthropic_messages;
pub mod audio_transcription;
pub mod chat;
pub(crate) mod ocr;

View file

@ -1,7 +1,6 @@
pub mod anthropic;
pub mod azure_ai;
pub mod base_llm;
pub mod bedrock;
pub(crate) mod cohere;
pub(crate) mod mistral;
pub mod openai;

View file

@ -3,9 +3,9 @@ use serde_json::{Map, Value};
use super::Error;
use crate::http_utils::string_headers as shared_string_headers;
pub(super) use crate::http_utils::{has_bearer_auth, has_header, truncate_error_body};
use crate::llms::anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG;
use crate::llms::azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG;
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use litellm_providers::anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG;
use litellm_providers::azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG;
use litellm_providers::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
const HEADER_CONTEXT: &str = "messages";

View file

@ -28,6 +28,22 @@ pub enum Error {
InvalidBedrockBase64(String),
}
impl From<litellm_providers::messages::Error> for Error {
fn from(error: litellm_providers::messages::Error) -> Self {
match error {
litellm_providers::messages::Error::MissingField(field) => Self::MissingField(field),
litellm_providers::messages::Error::InvalidRequest(message) => {
Self::InvalidRequest(message)
}
litellm_providers::messages::Error::InvalidResponse(message) => {
Self::InvalidResponse(message)
}
litellm_providers::messages::Error::Unsupported(reason) => Self::Unsupported(reason),
litellm_providers::messages::Error::Auth(error) => Self::Auth(error),
}
}
}
impl Error {
pub fn is_request(&self) -> bool {
match self {

View file

@ -40,6 +40,7 @@ pub(super) async fn execute_messages_provider_call(
request
.config
.transform_anthropic_messages_response(&request.model, response)
.map_err(Error::from)
}
pub(super) async fn execute_messages_provider_stream(

View file

@ -13,7 +13,7 @@ mod client;
mod common_utils;
mod handler;
mod prepare;
pub mod types;
pub use litellm_providers::messages::types;
use handler::{execute_messages_provider_call, execute_messages_provider_stream};
use types::{AnthropicMessagesResponse, MessagesRequest};

View file

@ -6,7 +6,7 @@ use super::types::{MessagesRequest, ProviderMessagesRequest};
use crate::litellm_core_utils::get_llm_provider_logic::{
CustomLlmProvider, get_custom_llm_provider,
};
use crate::llms::base_llm::anthropic_messages::transformation::{
use litellm_providers::base_llm::anthropic_messages::transformation::{
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
};

View file

@ -34,6 +34,14 @@ where
.then(|| "document".to_string()),
)
.collect();
let original_document =
serde_json::to_value(&request.document).map_err(|_| super::Error::RequestField {
path: "document".into(),
})?;
let prepared_document = composed
.get("document")
.filter(|prepared| **prepared != original_document)
.cloned();
let (body, headers) = if request.hooks.intercepts_requests() {
let changed = request
.hooks
@ -47,13 +55,19 @@ where
retained_fields,
})
.await?;
if !changed.body.is_object() {
let Value::Object(mut fields) = changed.body else {
return Err(super::Error::RequestField {
path: "guardrail.body".into(),
});
};
if let Some(prepared) =
prepared_document.filter(|_| fields.get("document") == Some(&original_document))
{
fields.insert("document".into(), prepared);
}
validate(&changed.body)?;
(changed.body, changed.headers)
let body = Value::Object(fields);
validate(&body)?;
(body, changed.headers)
} else {
(composed, headers.to_vec())
};

View file

@ -0,0 +1,16 @@
[package]
name = "litellm-providers"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
litellm-auth.workspace = true
litellm-auth-aws.workspace = true
serde.workspace = true
serde_json.workspace = true
thiserror.workspace = true
[dev-dependencies]
rstest.workspace = true

View file

@ -1,7 +1,7 @@
use serde_json::json;
use super::*;
use crate::chat_completions::Error;
use crate::chat::Error;
fn messages(value: Value) -> Vec<ChatMessage> {
serde_json::from_value(value).expect("valid messages")

View file

@ -1,19 +1,19 @@
use serde_json::{Map, Value, json};
use crate::chat_completions::Error;
use crate::chat_completions::conversation::{Conversation, build_conversation};
use crate::chat_completions::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat_completions::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
ProviderChatRequestData, ProviderChatResponseData,
};
use crate::constants::ANTHROPIC_OAUTH_TOKEN_PREFIX;
use crate::llms::anthropic::experimental_pass_through::messages::transformation::{
use crate::anthropic::ANTHROPIC_OAUTH_TOKEN_PREFIX;
use crate::anthropic::experimental_pass_through::messages::transformation::{
complete_anthropic_url, resolve_anthropic_api_key,
};
use crate::llms::base_llm::chat::transformation::{
use crate::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use crate::chat::Error;
use crate::chat::conversation::{Conversation, build_conversation};
use crate::chat::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse, ChatMessage,
ProviderChatRequestData, ProviderChatResponseData,
};
/// Anthropic parameter names, post `map_openai_params`, that the Rust path can
/// place verbatim in the Messages body.

View file

@ -1,4 +1,4 @@
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::messages::Error;
const ANTHROPIC_API_KEY_ENV: &str = "ANTHROPIC_API_KEY";

View file

@ -0,0 +1 @@
pub mod messages;

View file

@ -0,0 +1,4 @@
pub mod chat;
pub mod experimental_pass_through;
pub const ANTHROPIC_OAUTH_TOKEN_PREFIX: &str = "sk-ant-oat";

View file

@ -0,0 +1,31 @@
use thiserror::Error;
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("expected {expected}, got {actual}")]
InvalidType {
expected: &'static str,
actual: &'static str,
},
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub fn json_type_name(value: &serde_json::Value) -> &'static str {
match value {
serde_json::Value::Null => "null",
serde_json::Value::Bool(_) => "boolean",
serde_json::Value::Number(_) => "number",
serde_json::Value::String(_) => "string",
serde_json::Value::Array(_) => "array",
serde_json::Value::Object(_) => "object",
}
}
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::audio_transcription::transformation::{
use crate::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
@ -20,15 +20,15 @@ pub struct AudioTranscriptionRequest<'a> {
#[derive(Clone)]
pub struct ProviderAudioTranscriptionRequest {
pub(super) model: String,
pub(super) custom_llm_provider: String,
pub(super) config: &'static dyn BaseAudioTranscriptionConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) auth: AudioTranscriptionAuth,
pub(super) optional_params: Map<String, Value>,
pub(super) timeout: Option<Duration>,
pub model: String,
pub custom_llm_provider: String,
pub config: &'static dyn BaseAudioTranscriptionConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub auth: AudioTranscriptionAuth,
pub optional_params: Map<String, Value>,
pub timeout: Option<Duration>,
}
impl ProviderAudioTranscriptionRequest {

View file

@ -1,9 +1,9 @@
use serde_json::{Map, Value};
use crate::llms::anthropic::experimental_pass_through::messages::transformation::{
use crate::anthropic::experimental_pass_through::messages::transformation::{
ANTHROPIC_MESSAGES_CONFIG, AnthropicMessagesConfig, non_empty,
};
use crate::llms::base_llm::anthropic_messages::transformation::{
use crate::base_llm::anthropic_messages::transformation::{
BaseAnthropicMessagesConfig, MessagesAuthStrategy,
};
use crate::messages::Error;

View file

@ -0,0 +1 @@
pub mod anthropic;

View file

@ -0,0 +1 @@
pub mod transformation;

View file

@ -0,0 +1 @@
pub mod transformation;

View file

@ -1,7 +1,7 @@
use serde_json::{Map, Value};
use crate::chat_completions::Error;
use crate::chat_completions::types::{
use crate::chat::Error;
use crate::chat::types::{
ChatCompletionsResponse, ChatMessage, ChatMessageContent, ProviderChatRequestData,
ProviderChatResponseData,
};

View file

@ -0,0 +1,3 @@
pub mod anthropic_messages;
pub mod audio_transcription;
pub mod chat;

View file

@ -1,11 +1,11 @@
use serde_json::{Map, Value, json};
use crate::audio_transcription::Error;
use crate::audio_transcription::json_type_name;
use crate::audio_transcription::types::{
AudioTranscriptionRequestData, AudioTranscriptionResponseData,
};
use crate::http_utils::json_type_name;
use crate::llms::base_llm::audio_transcription::transformation::{
use crate::base_llm::audio_transcription::transformation::{
AudioTranscriptionAuth, BaseAudioTranscriptionConfig,
};
use litellm_auth_aws::constants::{BEDROCK_RUNTIME_ENDPOINT_TEMPLATE, BEDROCK_SERVICE};

View file

@ -1,16 +1,16 @@
use serde_json::{Map, Value, json};
use crate::chat_completions::Error;
use crate::chat_completions::conversation::{Conversation, TurnRole, build_conversation};
use crate::chat_completions::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat_completions::types::{
use crate::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use crate::chat::Error;
use crate::chat::conversation::{Conversation, TurnRole, build_conversation};
use crate::chat::response_utils::{finish_reason_for, unix_now, usage_from_parts};
use crate::chat::types::{
ChatCompletionsChoice, ChatCompletionsChoiceMessage, ChatCompletionsResponse,
ChatCompletionsUsage, ChatMessage, ChatMessageContent, ProviderChatRequestData,
ProviderChatResponseData,
};
use crate::llms::base_llm::chat::transformation::{
BaseConfig, ChatCompletionsAuth, Unsupported, unsupported_message, unsupported_param,
};
use litellm_auth_aws::constants::{AWS_BEARER_TOKEN_BEDROCK, BEDROCK_RUNTIME_ENDPOINT_TEMPLATE};
use litellm_auth_aws::{bedrock_model_id_and_region, resolve_bedrock_region};

View file

@ -1,7 +1,7 @@
use serde_json::json;
use super::*;
use crate::chat_completions::Error;
use crate::chat::Error;
fn messages(value: Value) -> Vec<ChatMessage> {
serde_json::from_value(value).expect("valid messages")

View file

@ -11,7 +11,7 @@
//! accepts; anything richer is declined upstream by the capability gate.
use super::types::{ChatMessage, ChatMessageContent};
use crate::constants::EMPTY_TEXT_PLACEHOLDER;
use crate::chat::EMPTY_TEXT_PLACEHOLDER;
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum TurnRole {

View file

@ -0,0 +1,21 @@
use thiserror::Error;
pub const EMPTY_TEXT_PLACEHOLDER: &str = " ";
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error("unsupported: {0}")]
Unsupported(&'static str),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub mod conversation;
pub mod response_utils;
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
use crate::base_llm::chat::transformation::{BaseConfig, ChatCompletionsAuth};
/// A `/chat/completions` call as it crosses into the core.
///
@ -22,26 +22,26 @@ pub struct ChatCompletionsRequest<'a> {
pub timeout: Option<Duration>,
}
pub(super) struct ResolvedChatCompletionsRequest<'a> {
pub(super) model: String,
pub(super) config: &'static dyn BaseConfig,
pub(super) messages: Vec<ChatMessage>,
pub(super) optional_params: Map<String, Value>,
pub(super) api_key: Option<&'a str>,
pub(super) api_base: Option<&'a str>,
pub(super) extra_headers: Option<Map<String, Value>>,
pub(super) timeout: Option<Duration>,
pub struct ResolvedChatCompletionsRequest<'a> {
pub model: String,
pub config: &'static dyn BaseConfig,
pub messages: Vec<ChatMessage>,
pub optional_params: Map<String, Value>,
pub api_key: Option<&'a str>,
pub api_base: Option<&'a str>,
pub extra_headers: Option<Map<String, Value>>,
pub timeout: Option<Duration>,
}
pub(super) struct ProviderChatCompletionsRequest {
pub(super) model: String,
pub(super) config: &'static dyn BaseConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) auth: ChatCompletionsAuth,
pub(super) optional_params: Map<String, Value>,
pub(super) timeout: Option<Duration>,
pub struct ProviderChatCompletionsRequest {
pub model: String,
pub config: &'static dyn BaseConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub auth: ChatCompletionsAuth,
pub optional_params: Map<String, Value>,
pub timeout: Option<Duration>,
}
/// The provider-shaped request body a config produces. Named rather than a bare

View file

@ -0,0 +1,8 @@
pub mod anthropic;
pub mod audio_transcription;
pub mod azure_ai;
pub mod base_llm;
pub mod bedrock;
pub mod chat;
pub mod messages;
pub mod provider_resolution;

View file

@ -0,0 +1,17 @@
use thiserror::Error;
#[derive(Clone, Debug, PartialEq, Eq, Error)]
pub enum Error {
#[error("missing required field: {0}")]
MissingField(&'static str),
#[error("invalid request: {0}")]
InvalidRequest(String),
#[error("invalid response: {0}")]
InvalidResponse(String),
#[error("unsupported: {0}")]
Unsupported(&'static str),
#[error(transparent)]
Auth(#[from] litellm_auth::Error),
}
pub mod types;

View file

@ -3,7 +3,7 @@ use std::time::Duration;
use serde::{Deserialize, Serialize};
use serde_json::{Map, Value};
use crate::llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
use crate::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig;
pub struct MessagesRequest<'a> {
pub model: &'a str,
@ -15,14 +15,14 @@ pub struct MessagesRequest<'a> {
pub timeout: Option<Duration>,
}
pub(super) struct ProviderMessagesRequest {
pub(super) provider: String,
pub(super) model: String,
pub(super) config: &'static dyn BaseAnthropicMessagesConfig,
pub(super) url: String,
pub(super) body: Value,
pub(super) upstream_headers: Vec<(String, String)>,
pub(super) timeout: Option<Duration>,
pub struct ProviderMessagesRequest {
pub provider: String,
pub model: String,
pub config: &'static dyn BaseAnthropicMessagesConfig,
pub url: String,
pub body: Value,
pub upstream_headers: Vec<(String, String)>,
pub timeout: Option<Duration>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]

View file

@ -0,0 +1,33 @@
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct CustomLlmProvider<'a> {
pub model: &'a str,
pub custom_llm_provider: &'a str,
}
pub fn get_custom_llm_provider<'a>(
model: &'a str,
custom_llm_provider: Option<&'a str>,
) -> Option<CustomLlmProvider<'a>> {
if let Some(custom_llm_provider) = custom_llm_provider.filter(|provider| !provider.is_empty()) {
return Some(CustomLlmProvider {
model: strip_custom_llm_provider_prefix(model, custom_llm_provider),
custom_llm_provider,
});
}
let (custom_llm_provider, model) = model.split_once('/')?;
if custom_llm_provider.is_empty() || model.is_empty() {
return None;
}
Some(CustomLlmProvider {
model,
custom_llm_provider,
})
}
fn strip_custom_llm_provider_prefix<'a>(model: &'a str, custom_llm_provider: &str) -> &'a str {
model
.strip_prefix(custom_llm_provider)
.and_then(|model| model.strip_prefix('/'))
.unwrap_or(model)
}

View file

@ -394,35 +394,20 @@ class LangFuseLogger:
status_message=status_message,
)
verbose_logger.debug("OUTPUT IN LANGFUSE: %s; original: %s", output, response_obj)
trace_id = None
generation_id = None
if self._is_langfuse_v2():
trace_id, generation_id = self._log_langfuse_v2(
user_id=user_id,
metadata=metadata,
litellm_params=litellm_params,
output=output,
start_time=start_time,
end_time=end_time,
kwargs=kwargs,
optional_params=optional_params,
input=input,
response_obj=response_obj,
level=level,
litellm_call_id=litellm_call_id,
)
elif response_obj is not None:
self._log_langfuse_v1(
user_id=user_id,
metadata=metadata,
output=output,
start_time=start_time,
end_time=end_time,
kwargs=kwargs,
optional_params=optional_params,
input=input,
response_obj=response_obj,
)
trace_id, generation_id = self._log_langfuse_v2(
user_id=user_id,
metadata=metadata,
litellm_params=litellm_params,
output=output,
start_time=start_time,
end_time=end_time,
kwargs=kwargs,
optional_params=optional_params,
input=input,
response_obj=response_obj,
level=level,
litellm_call_id=litellm_call_id,
)
verbose_logger.debug("Langfuse Layer Logging - final response object: %s", response_obj)
verbose_logger.info("Langfuse Layer Logging - logging success")
@ -518,58 +503,6 @@ class LangFuseLogger:
This approach does not impact latency and runs in the background
"""
def _is_langfuse_v2(self):
import langfuse
return Version(langfuse.version.__version__) >= Version("2.0.0")
def _log_langfuse_v1(
self,
user_id,
metadata,
output,
start_time,
end_time,
kwargs,
optional_params,
input,
response_obj,
):
from langfuse.model import CreateGeneration, CreateTrace
verbose_logger.warning(
"Please upgrade langfuse to v2.0.0 or higher: https://github.com/langfuse/langfuse-python/releases/tag/v2.0.1"
)
trace: Final = self.Langfuse.trace(
CreateTrace(
name=metadata.get("generation_name", "litellm-completion"),
input=input,
output=output,
userId=user_id,
)
)
custom_llm_provider: Final = cast(str | None, kwargs.get("custom_llm_provider"))
model_name: Final = reconstruct_model_name(kwargs.get("model", ""), custom_llm_provider, metadata)
trace.generation(
CreateGeneration(
name=metadata.get("generation_name", "litellm-completion"),
startTime=start_time,
endTime=end_time,
model=model_name,
modelParameters=optional_params,
prompt=input,
completion=output,
usage={
"prompt_tokens": response_obj.usage.prompt_tokens,
"completion_tokens": response_obj.usage.completion_tokens,
},
metadata=metadata,
)
)
def _log_langfuse_v2(
self,
user_id: str | None,

View file

@ -995,23 +995,6 @@ class PrometheusLogger(CustomLogger):
return label_filters
def _validate_configured_metric_labels(self, metric_name: str, labels: list[str]):
"""
Ensure that all the configured labels are valid for the metric
Raises ValueError if the metric labels are invalid and pretty prints the error
"""
label_error: Final = self._validate_single_metric_labels(metric_name, labels)
if label_error:
self._pretty_print_invalid_labels_error(
metric_name=label_error.metric_name,
invalid_labels=label_error.invalid_labels,
valid_labels=label_error.valid_labels,
)
raise ValueError(label_error.message)
return True
#########################################################
# Pretty print functions
#########################################################
@ -1090,108 +1073,10 @@ class PrometheusLogger(CustomLogger):
for label_error in validation_results.label_errors:
verbose_logger.error(label_error.message)
def _pretty_print_invalid_labels_error(
self, metric_name: str, invalid_labels: list[str], valid_labels: list[str]
) -> None:
"""Pretty print error message for invalid labels using rich"""
try:
from rich.console import Console
from rich.panel import Panel
from rich.table import Table
from rich.text import Text
console: Final = Console()
# Create error panel title
title: Final = Text(
f"🚨🚨 Invalid Labels for Metric: '{metric_name}'\nInvalid labels: {', '.join(invalid_labels)}\nPlease specify only valid labels below",
style="bold red",
)
# Create valid labels table
labels_table: Final = Table(
title="🏷️ Valid Labels for this Metric",
show_header=True,
header_style="bold green",
title_justify="left",
border_style="green",
)
labels_table.add_column("Valid Labels", style="cyan", no_wrap=True)
for label in sorted(valid_labels):
labels_table.add_row(label)
# Print everything in a nice panel
console.print("\n")
console.print(Panel(title, border_style="red"))
console.print(labels_table)
console.print("\n")
except ImportError:
# Fallback to simple logging if rich is not available
verbose_logger.error(
"Invalid labels for metric '%s': %s. Valid labels: %s",
metric_name,
invalid_labels,
sorted(valid_labels),
)
def _pretty_print_invalid_metric_error(self, invalid_metric_name: str, valid_metrics: tuple) -> None:
"""Pretty print error message for invalid metric name using rich"""
try:
from rich.console import Console
from rich.panel import Panel
from rich.table import Table
from rich.text import Text
console: Final = Console()
# Create error panel title
title: Final = Text(
f"🚨🚨 Invalid Metric Name: '{invalid_metric_name}'\nPlease specify one of the allowed metrics below",
style="bold red",
)
# Create valid metrics table
metrics_table: Final = Table(
title="📊 Valid Metric Names",
show_header=True,
header_style="bold green",
title_justify="left",
border_style="green",
)
metrics_table.add_column("Available Metrics", style="cyan", no_wrap=True)
for metric in sorted(valid_metrics):
metrics_table.add_row(metric)
# Print everything in a nice panel
console.print("\n")
console.print(Panel(title, border_style="red"))
console.print(metrics_table)
console.print("\n")
except ImportError:
# Fallback to simple logging if rich is not available
verbose_logger.error(
"Invalid metric name: %s. Valid metrics: %s", invalid_metric_name, sorted(valid_metrics)
)
#########################################################
# End of pretty print functions
#########################################################
def _valid_metric_name(self, metric_name: str):
"""
Raises ValueError if the metric name is invalid and pretty prints the error
"""
error: Final = self._validate_single_metric_name(metric_name)
if error:
self._pretty_print_invalid_metric_error(
invalid_metric_name=error.metric_name, valid_metrics=error.valid_metrics
)
raise ValueError(error.message)
def _pretty_print_prometheus_config(self, label_filters: dict[str, list[str]]) -> None:
"""Pretty print the processed prometheus configuration using rich"""
try:

View file

@ -90,6 +90,7 @@ from litellm.litellm_core_utils.logging_utils import (
truncate_base64_in_messages_async,
)
from litellm.litellm_core_utils.model_param_helper import ModelParamHelper
from litellm.litellm_core_utils.ptu_pricing import is_spilled_over_ptu_request
from litellm.litellm_core_utils.redact_messages import (
redact_message_input_output_from_custom_logger,
redact_message_input_output_from_logging,
@ -1746,8 +1747,14 @@ class Logging(LiteLLMLoggingBaseClass):
if transformed_result is not None:
result = transformed_result
result_hidden_params: Final = getattr(result, "_hidden_params", None) or MappingProxyType({})
result_additional_headers: Final = (
result_hidden_params.get("additional_headers")
if isinstance(result_hidden_params, dict)
else getattr(result_hidden_params, "additional_headers", None)
)
if isinstance(result, (BaseModel, HttpxBinaryResponseContent)) and hasattr(result, "_hidden_params"):
hidden_params: Final = getattr(result, "_hidden_params", {})
hidden_params: Final = result_hidden_params
if (
"response_cost" in hidden_params and hidden_params["response_cost"] is not None
): # use cost if already calculated
@ -1762,8 +1769,17 @@ class Logging(LiteLLMLoggingBaseClass):
router_model_id = self.get_router_model_id()
## RESPONSE COST ##
custom_pricing: Final = use_custom_pricing_for_model(
litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None)
spilled_over: Final = is_spilled_over_ptu_request(
model_info=_deployment_model_info(self.litellm_params if hasattr(self, "litellm_params") else None),
response_headers=self.model_call_details.get("response_headers"),
additional_headers=result_additional_headers,
)
custom_pricing: Final = (
False
if spilled_over
else use_custom_pricing_for_model(
litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None)
)
)
prompt = self._prompt_for_cost_calculation()
@ -5257,6 +5273,18 @@ def _get_custom_logger_settings_from_proxy_server(callback_name: str) -> dict:
return {}
def _deployment_model_info(litellm_params: dict | None) -> Mapping[str, object]:
"""The router-stamped deployment model_info from whichever metadata field carries it."""
if litellm_params is None:
return MappingProxyType({})
for metadata_key in ("metadata", "litellm_metadata"):
if not isinstance(metadata := litellm_params.get(metadata_key), Mapping):
continue
if model_info := metadata.get("model_info"):
return model_info
return MappingProxyType({})
def use_custom_pricing_for_model(litellm_params: dict | None) -> bool:
"""
Check if the model uses custom pricing

View file

@ -14,9 +14,11 @@ from typing import Final
from litellm.secret_managers.main import get_secret_bool
from litellm.types.router import ModelInfo
from litellm.types.utils import CustomPricingLiteLLMParams, MirroredPricingParams
from litellm.types.utils import AzureSpillover, CustomPricingLiteLLMParams, MirroredPricingParams
PTU_COST_ATTRIBUTION_ENV_VAR: Final = "LITELLM_ENABLE_PTU_COST_ATTRIBUTION"
AZURE_SPILLOVER_HEADER: Final = "x-ms-is-spilled-over"
AZURE_SPILLOVER_FROM_HEADER: Final = "x-ms-spillover-from-deployment"
def is_ptu_cost_attribution_enabled() -> bool:
@ -235,3 +237,33 @@ def zeroed_ptu_pricing(
),
}
)
def is_spilled_over_ptu_request(
model_info: Mapping[str, object],
response_headers: Mapping[str, object] | None,
additional_headers: Mapping[str, object] | None,
) -> bool:
"""Whether Azure served this request from pay-as-you-go capacity, so the zeroed PTU rates must not apply."""
if ptu_terms(model_info) is None:
return False
if not is_ptu_cost_attribution_enabled():
return False
return azure_spillover(response_headers, additional_headers) is not None
def azure_spillover(
response_headers: Mapping[str, object] | None,
additional_headers: Mapping[str, object] | None,
) -> AzureSpillover | None:
"""The spillover Azure reports in the response headers, else None."""
for headers, prefix in (
(response_headers, ""),
(additional_headers, "llm_provider-"),
):
if headers is None or str(headers.get(f"{prefix}{AZURE_SPILLOVER_HEADER}")).lower() != "true":
continue
return AzureSpillover(
from_deployment=str(v) if (v := headers.get(f"{prefix}{AZURE_SPILLOVER_FROM_HEADER}")) is not None else None
)
return None

View file

@ -113,14 +113,6 @@ class _PredibaseStreamData(TypedDict):
error: str | None
class _Ai21StreamData(TypedDict):
completions: Sequence[Mapping[str, Mapping[str, str]]]
class _MaritalkStreamData(TypedDict):
answer: str
class _NlpCloudStreamData(TypedDict):
generated_text: str
@ -129,25 +121,6 @@ class _AlephAlphaStreamData(TypedDict):
completions: Sequence[Mapping[str, str]]
class _AzureStreamChoice(TypedDict):
delta: Mapping[str, str] | None
finish_reason: str | None
class _AzureStreamData(TypedDict):
choices: Sequence[_AzureStreamChoice]
class _BasetenModelOutput(TypedDict):
data: NotRequired[Sequence[str]]
class _BasetenStreamData(TypedDict):
token: NotRequired[Mapping[str, str]]
model_output: NotRequired["_BasetenModelOutput | str"]
completion: NotRequired[object]
class _DeltaDumpDict(TypedDict):
role: NotRequired[str | None]
tool_calls: NotRequired[Sequence[Mapping[str, object]]]
@ -572,36 +545,6 @@ class CustomStreamWrapper:
except Exception as e:
raise e
def handle_ai21_chunk(self, chunk): # fake streaming
chunk = chunk.decode("utf-8")
data_json: Final[_Ai21StreamData] = json.loads(chunk)
try:
text: Final = data_json["completions"][0]["data"]["text"]
is_finished: Final = True
finish_reason: Final = "stop"
return {
"text": text,
"is_finished": is_finished,
"finish_reason": finish_reason,
}
except Exception:
raise ValueError(f"Unable to parse response. Original response: {chunk}")
def handle_maritalk_chunk(self, chunk): # fake streaming
chunk = chunk.decode("utf-8")
data_json: Final[_MaritalkStreamData] = json.loads(chunk)
try:
text: Final = data_json["answer"]
is_finished: Final = True
finish_reason: Final = "stop"
return {
"text": text,
"is_finished": is_finished,
"finish_reason": finish_reason,
}
except Exception:
raise ValueError(f"Unable to parse response. Original response: {chunk}")
def handle_nlp_cloud_chunk(self, chunk):
text = ""
is_finished = False
@ -640,46 +583,6 @@ class CustomStreamWrapper:
except Exception:
raise ValueError(f"Unable to parse response. Original response: {chunk}")
def handle_azure_chunk(self, chunk):
is_finished = False
finish_reason = ""
text = ""
print_verbose(f"chunk: {chunk}")
if "data: [DONE]" in chunk:
text = ""
is_finished = True
finish_reason = "stop"
return {
"text": text,
"is_finished": is_finished,
"finish_reason": finish_reason,
}
elif chunk.startswith("data:"):
data_json: Final[_AzureStreamData] = json.loads(chunk[5:]) # chunk.startswith("data:"):
try:
if len(data_json["choices"]) > 0:
delta: Final = data_json["choices"][0]["delta"]
text = "" if delta is None else delta.get("content", "")
if data_json["choices"][0].get("finish_reason", None):
is_finished = True
finish_reason = data_json["choices"][0]["finish_reason"]
print_verbose(f"text: {text}; is_finished: {is_finished}; finish_reason: {finish_reason}")
return {
"text": text,
"is_finished": is_finished,
"finish_reason": finish_reason,
}
except Exception:
raise ValueError(f"Unable to parse response. Original response: {chunk}")
elif "error" in chunk:
raise ValueError(f"Unable to parse response. Original response: {chunk}")
else:
return {
"text": text,
"is_finished": is_finished,
"finish_reason": finish_reason,
}
def handle_replicate_chunk(self, chunk):
try:
text = ""
@ -782,38 +685,6 @@ class CustomStreamWrapper:
except Exception as e:
raise e
def handle_baseten_chunk(self, chunk) -> str:
try:
chunk = chunk.decode("utf-8")
if len(chunk) > 0:
if chunk.startswith("data:"):
data_json: _BasetenStreamData = json.loads(chunk[5:])
if "token" in data_json and "text" in data_json["token"]:
return data_json["token"]["text"]
else:
return ""
data_json = json.loads(chunk)
if "model_output" in data_json:
if (
isinstance(data_json["model_output"], dict)
and "data" in data_json["model_output"]
and isinstance(data_json["model_output"]["data"], list)
):
return data_json["model_output"]["data"][0]
elif isinstance(data_json["model_output"], str):
return data_json["model_output"]
elif "completion" in data_json and isinstance(data_json["completion"], str):
return data_json["completion"]
else:
raise ValueError(f"Unable to parse response. Original response: {chunk}")
else:
return ""
else:
return ""
except Exception as e:
verbose_logger.exception("litellm.CustomStreamWrapper.handle_baseten_chunk(): Exception occured - %s", e)
return ""
def handle_triton_stream(self, chunk):
try:
if isinstance(chunk, dict):
@ -1305,18 +1176,6 @@ class CustomStreamWrapper:
completion_obj["content"] = response_obj["text"]
if response_obj["is_finished"]:
self.received_finish_reason = response_obj["finish_reason"]
elif self.custom_llm_provider and self.custom_llm_provider == "baseten": # baseten doesn't provide streaming
completion_obj["content"] = self.handle_baseten_chunk(chunk)
elif self.custom_llm_provider and self.custom_llm_provider == "ai21": # ai21 doesn't provide streaming
response_obj = self.handle_ai21_chunk(chunk)
completion_obj["content"] = response_obj["text"]
if response_obj["is_finished"]:
self.received_finish_reason = response_obj["finish_reason"]
elif self.custom_llm_provider and self.custom_llm_provider == "maritalk":
response_obj = self.handle_maritalk_chunk(chunk)
completion_obj["content"] = response_obj["text"]
if response_obj["is_finished"]:
self.received_finish_reason = response_obj["finish_reason"]
elif self.custom_llm_provider and self.custom_llm_provider == "vllm":
completion_obj["content"] = chunk[0].outputs[0].text
elif (
@ -1410,19 +1269,6 @@ class CustomStreamWrapper:
new_chunk = stream[:chunk_size]
completion_obj["content"] = new_chunk
self.completion_stream = stream[chunk_size:]
elif self.custom_llm_provider == "palm":
# fake streaming
response_obj = {}
if self.completion_stream is None or len(self.completion_stream) == 0:
if self.received_finish_reason is not None:
raise StopIteration
else:
self.received_finish_reason = "stop"
chunk_size = 30
stream = cast(Any, self.completion_stream)
new_chunk = stream[:chunk_size]
completion_obj["content"] = new_chunk
self.completion_stream = stream[chunk_size:]
elif self.custom_llm_provider == "triton":
response_obj = self.handle_triton_stream(chunk)
completion_obj["content"] = response_obj["text"]

View file

@ -561,6 +561,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM):
headers, response = self.make_sync_azure_openai_chat_completion_request(
azure_client=azure_client, data=data, timeout=timeout
)
logging_obj.model_call_details["response_headers"] = headers
streamwrapper: Final = CustomStreamWrapper(
completion_stream=response,
model=model,

View file

@ -18,6 +18,7 @@ from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation
)
from litellm.llms.bedrock.common_utils import (
apply_bedrock_invoke_structured_output,
bedrock_supports_tool_search,
get_anthropic_beta_from_headers,
normalize_bedrock_opus_output_config_effort,
normalize_custom_field_on_tools,
@ -265,7 +266,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
if tool_search_used and not (programmatic_tool_calling_used or input_examples_used):
beta_set.discard(ANTHROPIC_TOOL_SEARCH_BETA_HEADER)
if "opus-4" in model.lower() or "opus_4" in model.lower():
if bedrock_supports_tool_search(model):
beta_set.add("tool-search-tool-2025-10-19")
auto_beta_list: Final = filter_and_transform_beta_headers(

View file

@ -34,6 +34,7 @@ if TYPE_CHECKING:
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
def error_response_text(response: httpx.Response) -> str:
@ -878,9 +879,10 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
Whether Converse ``cachePoint`` blocks may be sent to this model.
Bedrock rejects requests carrying cachePoint blocks for models without prompt
caching support ("You invoked an unsupported model or your request did not allow
prompt caching"), so a model whose cost-map entry does not declare
OpenAI-family models only support implicit caching and never accept explicit
``cachePoint`` blocks. Bedrock rejects requests carrying cachePoint blocks for
models without prompt caching support ("You invoked an unsupported model or your
request did not allow prompt caching"), so a model whose cost-map entry does not declare
``supports_prompt_caching`` must not receive them. A model absent from the map
(an application inference profile ARN, a model newer than the map) keeps emitting
so existing caching setups never silently degrade. ``litellm.utils.supports_prompt_caching``
@ -888,6 +890,8 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
if model is None:
return True
if _OPENAI_FAMILY_MODEL_RE.search(model):
return False
entries: Final = tuple(
entry
for candidate in (model, get_bedrock_base_model(model))
@ -898,6 +902,20 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
return any(entry.get("supports_prompt_caching") is True for entry in entries)
def bedrock_supports_tool_search(model: str) -> bool:
"""
Whether Bedrock InvokeModel admits the ``tool_search_tool_*`` tool types on ``model``.
Backed by the ``supports_tool_search`` flag in ``model_prices_and_context_window.json``,
an exact entry or the ``claude-tool-search`` fallback rule for Claude 4.5 and newer, so a
newly released Claude carries the flag with no code change. An explicit ``false`` on the
resolved entry wins over the rule.
"""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
return AnthropicModelInfo._supports_model_capability(model, "supports_tool_search", "bedrock")
def is_claude_4_5_on_bedrock(model: str) -> bool:
"""
Check if the model supports Bedrock prompt caching with an extended '1h' TTL

View file

@ -31,6 +31,7 @@ from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation
from litellm.llms.bedrock.common_utils import (
BedrockError,
apply_bedrock_invoke_structured_output,
bedrock_supports_tool_search,
ensure_bedrock_anthropic_messages_tool_names,
get_anthropic_beta_from_headers,
is_claude_4_5_on_bedrock,
@ -386,9 +387,10 @@ class AmazonAnthropicClaudeMessagesConfig(
"""
Check if the model supports tool search on Bedrock.
The model map's ``supports_tool_search`` flag is authoritative when
``model`` resolves to an entry that sets it; the name patterns below
cover ids the map cannot resolve (ARNs, unlisted regional variants).
The model map's ``supports_tool_search`` flag is authoritative: an exact
entry, or the ``claude-tool-search`` fallback rule (Claude 4.5 and newer)
for ids the map cannot resolve (ARNs, unlisted regional variants) and for
mapped entries that carry no opinion.
Ref: https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool
@ -398,46 +400,7 @@ class AmazonAnthropicClaudeMessagesConfig(
Returns:
True if the model supports tool search on Bedrock
"""
catalog: Final = AnthropicModelInfo._get_provider_resolved_capability(model, "supports_tool_search", "bedrock")
if catalog is not None:
return catalog
model_lower: Final = model.lower()
supported_patterns: Final = [
# Opus 4.5
"opus-4.5",
"opus_4.5",
"opus-4-5",
"opus_4_5",
# Sonnet 4.5
"sonnet-4.5",
"sonnet_4.5",
"sonnet-4-5",
"sonnet_4_5",
# Opus 4.6
"opus-4.6",
"opus_4.6",
"opus-4-6",
"opus_4_6",
# sonnet 4.6
"sonnet-4.6",
"sonnet_4.6",
"sonnet-4-6",
"sonnet_4_6",
# Opus 4.7
"opus-4.7",
"opus_4.7",
"opus-4-7",
"opus_4_7",
# Haiku 4.5
"haiku-4.5",
"haiku_4.5",
"haiku-4-5",
"haiku_4_5",
]
return any(pattern in model_lower for pattern in supported_patterns)
return bedrock_supports_tool_search(model)
def _get_tool_search_beta_header_for_bedrock(
self,
@ -453,7 +416,8 @@ class AmazonAnthropicClaudeMessagesConfig(
Bedrock requires a different beta header for tool search than the
Anthropic API when tool search is used without programmatic tool
calling or input examples: `tool-search-tool-2025-10-19`, and only on
the models listed in `_supports_tool_search_on_bedrock`.
the models the model map flags as `supports_tool_search`
(`_supports_tool_search_on_bedrock`).
Ref: https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool

View file

@ -344,6 +344,10 @@ class BedrockMantleResponsesAPIConfig(BedrockMantleAuthMixin, OpenAIResponsesAPI
kept: Final = [item for item, _ in normalized if item is not None] # mutable-ok: ResponseInputParam is a list
return kept # pyright: ignore[reportReturnType] # Codex passthrough items sit outside the OpenAI input union
@staticmethod
def _model_map_lookup_name(model: str) -> str:
return model.split("/")[-1].removeprefix("openai.")
def map_openai_params(
self,
response_api_optional_params: ResponsesAPIOptionalRequestParams,

View file

@ -1,27 +1,6 @@
import copy
import time
import traceback
import types
from collections.abc import Callable
from typing import Final
import httpx
import litellm
from litellm.utils import Choices, Message, ModelResponse, Usage
class PalmError(Exception):
def __init__(self, status_code, message):
self.status_code = status_code
self.message = message
self.request = httpx.Request(
method="POST",
url="https://developers.generativeai.google/api/python/google/generativeai/chat",
)
self.response = httpx.Response(status_code=status_code, request=self.request)
super().__init__(self.message) # Call the base class constructor with the parameters it needs
class PalmConfig:
"""
@ -84,111 +63,3 @@ class PalmConfig:
)
and v is not None
}
def completion(
model: str,
messages: list,
model_response: ModelResponse,
print_verbose: Callable,
api_key,
encoding,
logging_obj,
optional_params: dict,
litellm_params=None,
logger_fn=None,
):
try:
import google.generativeai as palm
except Exception:
raise Exception("Importing google.generativeai failed, please run 'pip install -q google-generativeai")
palm.configure(api_key=api_key)
model = model
## Load Config
inference_params: Final = copy.deepcopy(optional_params)
inference_params.pop(
"stream", None
) # palm does not support streaming, so we handle this by fake streaming in main.py
config: Final = litellm.PalmConfig.get_config()
for k, v in config.items():
if (
k not in inference_params
): # completion(top_k=3) > palm_config(top_k=3) <- allows for dynamic variables to be passed in
inference_params[k] = v
prompt = ""
for message in messages:
if "role" in message:
if message["role"] == "user":
prompt += f"{message['content']}"
else:
prompt += f"{message['content']}"
else:
prompt += f"{message['content']}"
## LOGGING
logging_obj.pre_call(
input=prompt,
api_key="",
additional_args={"complete_input_dict": {"inference_params": inference_params}},
)
## COMPLETION CALL
try:
response: Final = palm.generate_text(prompt=prompt, **inference_params)
except Exception as e:
raise PalmError(
message=str(e),
status_code=500,
)
## LOGGING
logging_obj.post_call(
input=prompt,
api_key="",
original_response=response,
additional_args={"complete_input_dict": {}},
)
print_verbose(f"raw model_response: {response}")
## RESPONSE OBJECT
completion_response = response
try:
choices_list: Final = []
for idx, item in enumerate(completion_response.candidates):
if len(item["output"]) > 0:
message_obj = Message(content=item["output"])
else:
message_obj = Message(content=None)
choice_obj = Choices(index=idx + 1, message=message_obj)
choices_list.append(choice_obj)
model_response.choices = choices_list
except Exception:
raise PalmError(message=traceback.format_exc(), status_code=response.status_code)
try:
completion_response = model_response["choices"][0]["message"].get("content")
except Exception:
raise PalmError(
status_code=400,
message=f"No response received. Original response - {response}",
)
## CALCULATING USAGE - baseten charges on time, not tokens - have some mapping of cost here.
prompt_tokens: Final = len(encoding.encode(prompt))
completion_tokens: Final = len(encoding.encode(model_response["choices"][0]["message"].get("content", "")))
model_response.created = int(time.time())
model_response.model = "palm/" + model
usage: Final = Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=prompt_tokens + completion_tokens,
)
setattr(model_response, "usage", usage)
return model_response
def embedding():
# logic for parsing in - calling - parsing out model embedding calls
pass

View file

@ -38,7 +38,6 @@ def cost_per_token(
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
"""
## CALCULATE INPUT COST
return generic_cost_per_token(
model=model,
usage=usage,
@ -46,49 +45,6 @@ def cost_per_token(
service_tier=service_tier,
data_residency=data_residency,
)
# ### Non-cached text tokens
# non_cached_text_tokens = usage.prompt_tokens
# cached_tokens: Optional[int] = None
# if usage.prompt_tokens_details and usage.prompt_tokens_details.cached_tokens:
# cached_tokens = usage.prompt_tokens_details.cached_tokens
# non_cached_text_tokens = non_cached_text_tokens - cached_tokens
# prompt_cost: float = non_cached_text_tokens * model_info["input_cost_per_token"]
# ## Prompt Caching cost calculation
# if model_info.get("cache_read_input_token_cost") is not None and cached_tokens:
# # Note: We read ._cache_read_input_tokens from the Usage - since cost_calculator.py standardizes the cache read tokens on usage._cache_read_input_tokens
# prompt_cost += cached_tokens * (
# model_info.get("cache_read_input_token_cost", 0) or 0
# )
# _audio_tokens: Optional[int] = (
# usage.prompt_tokens_details.audio_tokens
# if usage.prompt_tokens_details is not None
# else None
# )
# _audio_cost_per_token: Optional[float] = model_info.get(
# "input_cost_per_audio_token"
# )
# if _audio_tokens is not None and _audio_cost_per_token is not None:
# audio_cost: float = _audio_tokens * _audio_cost_per_token
# prompt_cost += audio_cost
# ## CALCULATE OUTPUT COST
# completion_cost: float = (
# usage["completion_tokens"] * model_info["output_cost_per_token"]
# )
# _output_cost_per_audio_token: Optional[float] = model_info.get(
# "output_cost_per_audio_token"
# )
# _output_audio_tokens: Optional[int] = (
# usage.completion_tokens_details.audio_tokens
# if usage.completion_tokens_details is not None
# else None
# )
# if _output_cost_per_audio_token is not None and _output_audio_tokens is not None:
# audio_cost = _output_audio_tokens * _output_cost_per_audio_token
# completion_cost += audio_cost
# return prompt_cost, completion_cost
def cost_per_second(model: str, custom_llm_provider: str | None, duration: float = 0.0) -> tuple[float, float]:

View file

@ -125,6 +125,10 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
return False
return is_gpt_reasoning_series_name(model)
@staticmethod
def _model_map_lookup_name(model: str) -> str:
return model
@staticmethod
def _supports_reasoning_effort_none(model: str) -> bool:
"""Return True if the model supports reasoning.effort='none'."""
@ -208,8 +212,9 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
) -> dict:
"""No mapping applied since inputs are in OpenAI spec already.
GPT-5 models have restrictions on temperature (only temperature=1
is accepted unless reasoning_effort='none' on models that support it).
GPT-5 models have restrictions on temperature and top_p (only temperature=1
is accepted, and top_p is rejected, unless reasoning.effort resolves to
'none' on models that support it).
Apply the same validation used by the chat completions path.
"""
params: Final = dict(response_api_optional_params)
@ -234,13 +239,16 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
status_code=400,
)
if self._is_gpt_5_model(model=model):
lookup_name: Final = self._model_map_lookup_name(model)
if self._is_gpt_5_model(model=lookup_name):
reasoning: Final = params.get("reasoning") or {}
effort: Final = reasoning.get("effort") if isinstance(reasoning, dict) else None
supports_none: Final = self._supports_reasoning_effort_none(model=lookup_name)
effort_is_none: Final = supports_none and self._effort_resolves_to_none(lookup_name, effort)
temperature: Final = params.get("temperature")
if temperature is not None and temperature != 1:
reasoning: Final = params.get("reasoning") or {}
effort: Final = reasoning.get("effort") if isinstance(reasoning, dict) else None
supports_none: Final = self._supports_reasoning_effort_none(model=model)
if supports_none and self._effort_resolves_to_none(model, effort):
if effort_is_none:
pass # flexible temperature allowed
elif drop_params or litellm.drop_params:
params.pop("temperature", None)
@ -256,6 +264,20 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
status_code=400,
)
if "top_p" in params and not effort_is_none:
if drop_params or litellm.drop_params:
params.pop("top_p", None)
else:
raise litellm.UnsupportedParamsError(
message=(
f"{model} only supports top_p when reasoning.effort resolves to 'none', "
"either set explicitly on the request or declared as the model's "
"default_reasoning_effort. "
"To drop unsupported params set `litellm.drop_params = True`"
),
status_code=400,
)
return params
def transform_responses_api_request(

View file

@ -20,7 +20,6 @@ class VertexAITokenCounter(GoogleAIStudioTokenCounter, VertexBase):
vertex_credentials: Final = self.get_vertex_ai_credentials(litellm_params=litellm_params)
vertex_project = self.get_vertex_ai_project(litellm_params=litellm_params)
vertex_location: Final = self.get_vertex_ai_location(litellm_params=litellm_params)
should_use_v1beta1_features: Final = self.is_using_v1beta1_features(litellm_params)
_auth_header, vertex_project = await self._ensure_access_token_async(
credentials=vertex_credentials,
project_id=vertex_project,
@ -37,7 +36,6 @@ class VertexAITokenCounter(GoogleAIStudioTokenCounter, VertexBase):
stream=False,
custom_llm_provider="vertex_ai",
api_base=None,
should_use_v1beta1_features=should_use_v1beta1_features,
mode="count_tokens",
)
headers = {

View file

@ -2701,8 +2701,6 @@ class VertexLLM(VertexBase):
gemini_api_key: str | None = None,
extra_headers: dict | None = None,
) -> CustomStreamWrapper:
should_use_v1beta1_features: Final = self.is_using_v1beta1_features(optional_params=optional_params)
_auth_header, vertex_project = await self._ensure_access_token_async(
credentials=vertex_credentials,
project_id=vertex_project,
@ -2722,7 +2720,6 @@ class VertexLLM(VertexBase):
stream=stream,
custom_llm_provider=custom_llm_provider,
api_base=api_base,
should_use_v1beta1_features=should_use_v1beta1_features,
use_psc_endpoint_format=use_psc_endpoint_format,
)
@ -2797,8 +2794,6 @@ class VertexLLM(VertexBase):
gemini_api_key: str | None = None,
extra_headers: dict | None = None,
) -> ModelResponse | CustomStreamWrapper:
should_use_v1beta1_features: Final = self.is_using_v1beta1_features(optional_params=optional_params)
_auth_header, vertex_project = await self._ensure_access_token_async(
credentials=vertex_credentials,
project_id=vertex_project,
@ -2818,7 +2813,6 @@ class VertexLLM(VertexBase):
stream=stream,
custom_llm_provider=custom_llm_provider,
api_base=api_base,
should_use_v1beta1_features=should_use_v1beta1_features,
use_psc_endpoint_format=use_psc_endpoint_format,
)
@ -2981,8 +2975,6 @@ class VertexLLM(VertexBase):
extra_headers=extra_headers,
)
should_use_v1beta1_features: Final = self.is_using_v1beta1_features(optional_params=optional_params)
_auth_header, vertex_project = self._ensure_access_token(
credentials=vertex_credentials,
project_id=vertex_project,
@ -3002,7 +2994,6 @@ class VertexLLM(VertexBase):
stream=stream,
custom_llm_provider=custom_llm_provider,
api_base=api_base,
should_use_v1beta1_features=should_use_v1beta1_features,
use_psc_endpoint_format=use_psc_endpoint_format,
)
headers: Final = VertexGeminiConfig().validate_environment(

View file

@ -65,8 +65,6 @@ class VertexEmbedding(VertexBase):
litellm_params=litellm_params,
)
should_use_v1beta1_features: Final = self.is_using_v1beta1_features(optional_params=optional_params)
_auth_header, vertex_project = self._ensure_access_token(
credentials=vertex_credentials,
project_id=vertex_project,
@ -85,7 +83,6 @@ class VertexEmbedding(VertexBase):
stream=False,
custom_llm_provider=custom_llm_provider,
api_base=api_base,
should_use_v1beta1_features=should_use_v1beta1_features,
mode="embedding",
use_psc_endpoint_format=use_psc_endpoint_format,
)
@ -160,7 +157,6 @@ class VertexEmbedding(VertexBase):
"""
Async embedding implementation
"""
should_use_v1beta1_features: Final = self.is_using_v1beta1_features(optional_params=optional_params)
_auth_header, vertex_project = await self._ensure_access_token_async(
credentials=vertex_credentials,
project_id=vertex_project,
@ -179,7 +175,6 @@ class VertexEmbedding(VertexBase):
stream=False,
custom_llm_provider=custom_llm_provider,
api_base=api_base,
should_use_v1beta1_features=should_use_v1beta1_features,
mode="embedding",
use_psc_endpoint_format=use_psc_endpoint_format,
)

View file

@ -618,15 +618,6 @@ class VertexBase:
project_id=project_id,
)
def is_using_v1beta1_features(self, optional_params: dict) -> bool:
"""
use this helper to decide if request should be sent to v1 or v1beta1
Returns true if any beta feature is enabled
Returns false in all other cases
"""
return False
def _check_custom_proxy(
self,
api_base: str | None,

View file

@ -206,7 +206,7 @@ from .llms.custom_httpx.aiohttp_handler import BaseLLMAIOHTTPHandler
from .llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
from .llms.custom_llm import CustomLLM, custom_chat_llm_router
from .llms.databricks.embed.handler import DatabricksEmbeddingHandler
from .llms.deprecated_providers import aleph_alpha, palm
from .llms.deprecated_providers import aleph_alpha
from .llms.gdc.chat.transformation import GDCGeminiConfig
from .llms.gemini.common_utils import get_api_key_from_env
from .llms.groq.chat.handler import GroqChatCompletion

View file

@ -1810,6 +1810,7 @@
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -1847,6 +1848,7 @@
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -1884,6 +1886,7 @@
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 5.5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -1921,6 +1924,7 @@
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 5.5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -1957,6 +1961,7 @@
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 5.5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -1993,6 +1998,7 @@
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 5.5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2029,6 +2035,7 @@
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2067,6 +2074,7 @@
"cache_read_input_token_cost": 5e-07,
"input_cost_per_token": 5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2105,6 +2113,7 @@
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 5.5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2143,6 +2152,7 @@
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 5.5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2180,6 +2190,7 @@
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 5.5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2217,6 +2228,7 @@
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 5.5e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2287,6 +2299,7 @@
"cache_read_input_token_cost": 2e-07,
"input_cost_per_token": 2e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2325,6 +2338,7 @@
"cache_read_input_token_cost": 2e-07,
"input_cost_per_token": 2e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2363,6 +2377,7 @@
"cache_read_input_token_cost": 2.2e-07,
"input_cost_per_token": 2.2e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2401,6 +2416,7 @@
"cache_read_input_token_cost": 2.2e-07,
"input_cost_per_token": 2.2e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2438,6 +2454,7 @@
"cache_read_input_token_cost": 2.2e-07,
"input_cost_per_token": 2.2e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -2475,6 +2492,7 @@
"cache_read_input_token_cost": 2.2e-07,
"input_cost_per_token": 2.2e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -5282,7 +5300,7 @@
"supports_web_search": false
},
"azure/gpt-4.1-nano": {
"deprecation_date": "2027-04-14",
"deprecation_date": "2026-10-14",
"cache_read_input_token_cost": 2.5e-08,
"input_cost_per_token": 1e-07,
"input_cost_per_token_batches": 5e-08,
@ -5316,7 +5334,7 @@
"supports_vision": true
},
"azure/gpt-4.1-nano-2025-04-14": {
"deprecation_date": "2027-04-14",
"deprecation_date": "2026-10-14",
"cache_read_input_token_cost": 2.5e-08,
"input_cost_per_token": 1e-07,
"input_cost_per_token_batches": 5e-08,
@ -9473,7 +9491,7 @@
]
},
"azure/gpt-image-1.5": {
"deprecation_date": "2027-06-16",
"deprecation_date": "2026-12-16",
"cache_read_input_token_cost": 1.25e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_image_token": 8e-06,
@ -9487,7 +9505,7 @@
},
"azure/gpt-image-1.5-2025-12-16": {
"cache_read_input_token_cost": 1.25e-06,
"deprecation_date": "2027-06-16",
"deprecation_date": "2026-12-16",
"input_cost_per_token": 5e-06,
"input_cost_per_image_token": 8e-06,
"litellm_provider": "azure",
@ -10189,7 +10207,7 @@
"supports_web_search": false
},
"azure/us/gpt-4.1-nano-2025-04-14": {
"deprecation_date": "2027-04-14",
"deprecation_date": "2026-10-14",
"cache_read_input_token_cost": 2.8e-08,
"input_cost_per_token": 1.1e-07,
"input_cost_per_token_batches": 5.5e-08,
@ -23788,7 +23806,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
"supports_vision": true
},
"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf": {
"input_cost_per_token": 1.2e-06,
@ -24114,7 +24132,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
"supports_vision": true
},
"fireworks_ai/qwen3p7-plus": {
"cache_read_input_token_cost": 8e-08,
@ -46130,6 +46148,7 @@
"cache_read_input_token_cost": 2.4e-07,
"input_cost_per_token": 2.4e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -46163,6 +46182,7 @@
"cache_read_input_token_cost": 6e-07,
"input_cost_per_token": 6e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -46195,6 +46215,7 @@
"cache_read_input_token_cost": 6e-07,
"input_cost_per_token": 6e-06,
"litellm_provider": "bedrock_converse",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -57941,7 +57962,7 @@
"output_cost_per_token": 2.5e-06,
"cache_read_input_token_cost": 2e-07,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_input_tokens": 1048576,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
@ -59505,6 +59526,15 @@
"supports_mid_conversation_system": true
}
},
{
"name": "claude-tool-search",
"pattern": "claude-[a-z]+-(?:4[-._](?:[5-9]|[1-9]\\d)(?!\\d)|[5-9](?!\\d)(?:[-._]\\d{1,2}(?!\\d))?)",
"fill_missing_for_providers": ["anthropic", "bedrock", "bedrock_converse", "vertex_ai-anthropic_models"],
"description": "Claude at version 4.5 or higher, in any id shape that contains claude-<family>-: minors 4.5 through 4.99, any later major-minor, and bare 5+ majors so a new family like claude-fable-5 matches. Two-digit majors are deliberately not matched so ids like claude-opus-41 (4.1) are not read as major 41. Anthropic's tool search docs list every Claude 4.5 and newer model as supported and Opus 4.1 and earlier as unsupported, so the flag follows the version instead of a per-model list. azure_ai is left out on purpose: Anthropic documents tool search as unavailable on Azure-hosted Foundry deployments, and the azure_ai/ key cannot tell those from Anthropic-hosted ones.",
"model_info": {
"supports_tool_search": true
}
},
{
"name": "wandb-reasoning-baseline",
"pattern": "^wandb/",
@ -63253,6 +63283,7 @@
"cache_read_input_token_cost": 2.4e-07,
"input_cost_per_token": 2.4e-06,
"litellm_provider": "bedrock",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -63285,6 +63316,7 @@
"cache_read_input_token_cost": 6e-07,
"input_cost_per_token": 6e-06,
"litellm_provider": "bedrock",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -63316,6 +63348,7 @@
"cache_read_input_token_cost": 6e-07,
"input_cost_per_token": 6e-06,
"litellm_provider": "bedrock",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -63456,6 +63489,7 @@
"cache_read_input_token_cost": 2.4e-07,
"input_cost_per_token": 2.4e-06,
"litellm_provider": "bedrock",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -63488,6 +63522,7 @@
"cache_read_input_token_cost": 6e-07,
"input_cost_per_token": 6e-06,
"litellm_provider": "bedrock",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -63519,6 +63554,7 @@
"cache_read_input_token_cost": 6e-07,
"input_cost_per_token": 6e-06,
"litellm_provider": "bedrock",
"supports_tool_search": true,
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
@ -63678,7 +63714,7 @@
"bedrock_mantle/us-gov-west-1/xai.grok-4.3": {
"use_openai_responses_path": true,
"litellm_provider": "bedrock_mantle",
"max_input_tokens": 131072,
"max_input_tokens": 1048576,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
@ -67475,6 +67511,7 @@
"source": "https://api.together.ai/v1/models"
},
"azure/eu/codex-mini": {
"deprecation_date": "2026-11-15",
"cache_read_input_token_cost": 4.13e-07,
"input_cost_per_token": 1.65e-06,
"litellm_provider": "azure",
@ -67490,6 +67527,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-4.1": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_priority": 9.63e-07,
"input_cost_per_token": 2.2e-06,
@ -67503,6 +67541,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-4.1-mini": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 1.1e-07,
"cache_read_input_token_cost_priority": 1.93e-07,
"input_cost_per_token": 4.4e-07,
@ -67516,6 +67555,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-4.1-nano": {
"deprecation_date": "2026-10-14",
"cache_read_input_token_cost": 2.8e-08,
"input_cost_per_token": 1.1e-07,
"input_cost_per_token_batches": 5.5e-08,
@ -67526,6 +67566,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-4o-2024-05-13": {
"deprecation_date": "2026-10-01",
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_batches": 2.75e-06,
"litellm_provider": "azure",
@ -67535,6 +67576,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5": {
"deprecation_date": "2027-02-09",
"cache_read_input_token_cost": 1.375e-07,
"cache_read_input_token_cost_priority": 2.75e-07,
"input_cost_per_token": 1.375e-06,
@ -67548,6 +67590,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5-codex": {
"deprecation_date": "2027-03-17",
"cache_read_input_token_cost": 1.38e-07,
"input_cost_per_token": 1.375e-06,
"litellm_provider": "azure",
@ -67556,6 +67599,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5-mini": {
"deprecation_date": "2027-02-09",
"cache_read_input_token_cost": 2.75e-08,
"cache_read_input_token_cost_priority": 4.95e-08,
"input_cost_per_token": 2.75e-07,
@ -67569,6 +67613,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5-nano": {
"deprecation_date": "2027-02-09",
"cache_read_input_token_cost": 5.5e-09,
"input_cost_per_token": 5.5e-08,
"input_cost_per_token_batches": 2.75e-08,
@ -67579,6 +67624,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5-pro": {
"deprecation_date": "2027-04-07",
"input_cost_per_token": 1.65e-05,
"input_cost_per_token_batches": 8.25e-06,
"litellm_provider": "azure",
@ -67588,6 +67634,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.1-codex-max": {
"deprecation_date": "2027-05-18",
"cache_read_input_token_cost": 1.375e-07,
"input_cost_per_token": 1.375e-06,
"litellm_provider": "azure",
@ -67596,6 +67643,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.2": {
"deprecation_date": "2027-06-08",
"cache_read_input_token_cost": 1.925e-07,
"cache_read_input_token_cost_priority": 3.85e-07,
"input_cost_per_token": 1.925e-06,
@ -67609,6 +67657,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.2-chat": {
"deprecation_date": "2026-06-29",
"cache_read_input_token_cost": 1.925e-07,
"input_cost_per_token": 1.925e-06,
"litellm_provider": "azure",
@ -67617,6 +67666,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.2-codex": {
"deprecation_date": "2027-07-13",
"cache_read_input_token_cost": 1.925e-07,
"input_cost_per_token": 1.925e-06,
"litellm_provider": "azure",
@ -67634,6 +67684,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.3-chat": {
"deprecation_date": "2026-06-29",
"cache_read_input_token_cost": 1.925e-07,
"input_cost_per_token": 1.925e-06,
"litellm_provider": "azure",
@ -67642,6 +67693,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.3-codex": {
"deprecation_date": "2027-08-24",
"cache_read_input_token_cost": 1.925e-07,
"cache_read_input_token_cost_priority": 3.85e-07,
"input_cost_per_token": 1.925e-06,
@ -67653,6 +67705,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.4-mini": {
"deprecation_date": "2027-09-21",
"cache_read_input_token_cost": 8.25e-08,
"cache_read_input_token_cost_priority": 1.65e-07,
"input_cost_per_token": 8.25e-07,
@ -67666,6 +67719,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.4-nano": {
"deprecation_date": "2027-09-21",
"cache_read_input_token_cost": 2.2e-08,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_batches": 1.1e-07,
@ -67676,6 +67730,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/gpt-5.4-pro": {
"deprecation_date": "2027-09-07",
"input_cost_per_token": 3.3e-05,
"input_cost_per_token_above_272k_tokens": 6.6e-05,
"input_cost_per_token_batches": 1.65e-05,
@ -67718,6 +67773,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/o3-2025-04-16": {
"deprecation_date": "2026-11-19",
"cache_read_input_token_cost": 5.5e-07,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_batches": 1.1e-06,
@ -67728,6 +67784,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/o3-deep-research": {
"deprecation_date": "2026-11-19",
"cache_read_input_token_cost": 2.75e-06,
"input_cost_per_token": 1.1e-05,
"litellm_provider": "azure",
@ -67736,6 +67793,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/o4-mini-2025-04-16": {
"deprecation_date": "2026-11-19",
"cache_read_input_token_cost": 3.03e-07,
"input_cost_per_token": 1.21e-06,
"input_cost_per_token_batches": 6.05e-07,
@ -67746,18 +67804,21 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/text-embedding-3-large": {
"deprecation_date": "2028-02-09",
"input_cost_per_token": 1.43e-07,
"litellm_provider": "azure",
"mode": "embedding",
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/text-embedding-3-small": {
"deprecation_date": "2028-02-09",
"input_cost_per_token": 2.2e-08,
"litellm_provider": "azure",
"mode": "embedding",
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/eu/text-embedding-ada-002": {
"deprecation_date": "2028-02-09",
"input_cost_per_token": 1.1e-07,
"litellm_provider": "azure",
"mode": "embedding",
@ -67804,6 +67865,7 @@
"supports_web_search": true
},
"azure/us/codex-mini": {
"deprecation_date": "2026-11-15",
"cache_read_input_token_cost": 4.13e-07,
"input_cost_per_token": 1.65e-06,
"litellm_provider": "azure",
@ -67819,6 +67881,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-4.1": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_priority": 9.63e-07,
"input_cost_per_token": 2.2e-06,
@ -67832,6 +67895,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-4.1-mini": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 1.1e-07,
"cache_read_input_token_cost_priority": 1.93e-07,
"input_cost_per_token": 4.4e-07,
@ -67845,6 +67909,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-4.1-nano": {
"deprecation_date": "2026-10-14",
"cache_read_input_token_cost": 2.8e-08,
"input_cost_per_token": 1.1e-07,
"input_cost_per_token_batches": 5.5e-08,
@ -67855,6 +67920,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-4o-2024-05-13": {
"deprecation_date": "2026-10-01",
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_batches": 2.75e-06,
"litellm_provider": "azure",
@ -67864,6 +67930,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5": {
"deprecation_date": "2027-02-09",
"cache_read_input_token_cost": 1.375e-07,
"cache_read_input_token_cost_priority": 2.75e-07,
"input_cost_per_token": 1.375e-06,
@ -67877,6 +67944,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5-codex": {
"deprecation_date": "2027-03-17",
"cache_read_input_token_cost": 1.38e-07,
"input_cost_per_token": 1.375e-06,
"litellm_provider": "azure",
@ -67885,6 +67953,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5-mini": {
"deprecation_date": "2027-02-09",
"cache_read_input_token_cost": 2.75e-08,
"cache_read_input_token_cost_priority": 4.95e-08,
"input_cost_per_token": 2.75e-07,
@ -67898,6 +67967,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5-nano": {
"deprecation_date": "2027-02-09",
"cache_read_input_token_cost": 5.5e-09,
"input_cost_per_token": 5.5e-08,
"input_cost_per_token_batches": 2.75e-08,
@ -67908,6 +67978,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5-pro": {
"deprecation_date": "2027-04-07",
"input_cost_per_token": 1.65e-05,
"input_cost_per_token_batches": 8.25e-06,
"litellm_provider": "azure",
@ -67917,6 +67988,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.1-codex-max": {
"deprecation_date": "2027-05-18",
"cache_read_input_token_cost": 1.375e-07,
"input_cost_per_token": 1.375e-06,
"litellm_provider": "azure",
@ -67925,6 +67997,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.2": {
"deprecation_date": "2027-06-08",
"cache_read_input_token_cost": 1.925e-07,
"cache_read_input_token_cost_priority": 3.85e-07,
"input_cost_per_token": 1.925e-06,
@ -67938,6 +68011,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.2-chat": {
"deprecation_date": "2026-06-29",
"cache_read_input_token_cost": 1.925e-07,
"input_cost_per_token": 1.925e-06,
"litellm_provider": "azure",
@ -67946,6 +68020,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.2-codex": {
"deprecation_date": "2027-07-13",
"cache_read_input_token_cost": 1.925e-07,
"input_cost_per_token": 1.925e-06,
"litellm_provider": "azure",
@ -67963,6 +68038,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.3-chat": {
"deprecation_date": "2026-06-29",
"cache_read_input_token_cost": 1.925e-07,
"input_cost_per_token": 1.925e-06,
"litellm_provider": "azure",
@ -67971,6 +68047,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.3-codex": {
"deprecation_date": "2027-08-24",
"cache_read_input_token_cost": 1.925e-07,
"cache_read_input_token_cost_priority": 3.85e-07,
"input_cost_per_token": 1.925e-06,
@ -67982,6 +68059,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.4-mini": {
"deprecation_date": "2027-09-21",
"cache_read_input_token_cost": 8.25e-08,
"cache_read_input_token_cost_priority": 1.65e-07,
"input_cost_per_token": 8.25e-07,
@ -67995,6 +68073,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.4-nano": {
"deprecation_date": "2027-09-21",
"cache_read_input_token_cost": 2.2e-08,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_batches": 1.1e-07,
@ -68005,6 +68084,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/gpt-5.4-pro": {
"deprecation_date": "2027-09-07",
"input_cost_per_token": 3.3e-05,
"input_cost_per_token_above_272k_tokens": 6.6e-05,
"input_cost_per_token_batches": 1.65e-05,
@ -68034,6 +68114,7 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/o3-deep-research": {
"deprecation_date": "2026-11-19",
"cache_read_input_token_cost": 2.75e-06,
"input_cost_per_token": 1.1e-05,
"litellm_provider": "azure",
@ -68042,18 +68123,21 @@
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/text-embedding-3-large": {
"deprecation_date": "2028-02-09",
"input_cost_per_token": 1.43e-07,
"litellm_provider": "azure",
"mode": "embedding",
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/text-embedding-3-small": {
"deprecation_date": "2028-02-09",
"input_cost_per_token": 2.2e-08,
"litellm_provider": "azure",
"mode": "embedding",
"source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'"
},
"azure/us/text-embedding-ada-002": {
"deprecation_date": "2028-02-09",
"input_cost_per_token": 1.1e-07,
"litellm_provider": "azure",
"mode": "embedding",
@ -69184,6 +69268,27 @@
"supports_reasoning": true,
"supports_vision": true
},
"typesafe/jev-1.13.0": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-latest": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"typesafe/jev-preview": {
"input_cost_per_token": 4.2e-08,
"litellm_provider": "typesafe",
"mode": "evaluation",
"output_cost_per_token": 0.0,
"source": "https://docs.typesafe.ai/models"
},
"wandb/zai-org/GLM-5.3-Flash": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_token": 1.5e-07,

View file

@ -208,6 +208,7 @@ LAZY_FEATURES: Final[tuple[LazyFeature, ...]] = (
"/nvidia_nim/",
"/openai/",
"/openai_passthrough/",
"/typesafe/",
"/vertex-ai/",
"/vertex_ai/",
"/vllm/",

View file

@ -20373,6 +20373,228 @@
]
}
},
"/typesafe/{endpoint}": {
"delete": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__delete",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
},
"get": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__get",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
},
"patch": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__patch",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
},
"post": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__post",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
},
"put": {
"description": "[Docs](https://docs.litellm.ai/docs/pass_through/typesafe)",
"operationId": "typesafe_proxy_route_typesafe__endpoint__put",
"parameters": [
{
"in": "path",
"name": "endpoint",
"required": true,
"schema": {
"title": "Endpoint",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Typesafe Proxy Route",
"tags": [
"llm_passthrough"
]
}
},
"/vertex_ai/discovery/{endpoint}": {
"delete": {
"description": "Call any vertex discovery endpoint using the proxy.\n\nJust use `{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`\n\nTarget url: `https://discoveryengine.googleapis.com`",
@ -34364,6 +34586,20 @@
"title": "Policy Name",
"type": "string"
},
"priority": {
"anyOf": [
{
"maximum": 2147483647.0,
"minimum": -2147483648.0,
"type": "integer"
},
{
"type": "null"
}
],
"description": "Explicit execution order, lower runs first. Prioritised attachments run before those without one.",
"title": "Priority"
},
"scope": {
"anyOf": [
{
@ -34477,6 +34713,18 @@
"title": "Policy Name",
"type": "string"
},
"priority": {
"anyOf": [
{
"type": "integer"
},
{
"type": "null"
}
],
"description": "Explicit execution order, lower runs first. Prioritised attachments run before those without one.",
"title": "Priority"
},
"scope": {
"anyOf": [
{
@ -36497,6 +36745,20 @@
"title": "Policy Name",
"type": "string"
},
"priority": {
"anyOf": [
{
"maximum": 2147483647.0,
"minimum": -2147483648.0,
"type": "integer"
},
{
"type": "null"
}
],
"description": "Explicit execution order, lower runs first. Prioritised attachments run before those without one.",
"title": "Priority"
},
"scope": {
"anyOf": [
{

View file

@ -1,41 +0,0 @@
### DEPRECATED ###
## unused file. initially written for json logging on proxy.
import json
import logging
import os
from logging import Formatter
from typing import Final
from litellm import json_logs
# Set default log level to INFO
log_level: Final = os.getenv("LITELLM_LOG", "INFO")
numeric_level: Final[str] = getattr(logging, log_level.upper())
class JsonFormatter(Formatter):
def __init__(self):
super().__init__()
def format(self, record):
json_record: Final = {
"message": record.getMessage(),
"level": record.levelname,
"timestamp": self.formatTime(record, self.datefmt),
}
return json.dumps(json_record)
logger: Final = logging.root
handler: Final = logging.StreamHandler()
if json_logs:
handler.setFormatter(JsonFormatter())
else:
formatter: Final = logging.Formatter(
"\033[92m%(asctime)s - %(name)s:%(levelname)s\033[0m: %(filename)s:%(lineno)s - %(message)s",
datefmt="%H:%M:%S",
)
handler.setFormatter(formatter)
logger.handlers = [handler]
logger.setLevel(numeric_level)

View file

@ -51,6 +51,7 @@ from litellm.types.router import RouterErrors, UpdateRouterConfig
from litellm.types.router_weights import validate_router_settings_dict
from litellm.types.secret_managers.main import KeyManagementSystem
from litellm.types.utils import (
AzureSpillover,
CallTypes,
CostBreakdown,
EmbeddingResponse,
@ -483,6 +484,7 @@ class LiteLLMRoutes(enum.Enum):
"/eu.assemblyai",
"/vllm",
"/mistral",
"/typesafe",
"/milvus",
"/gigachat",
"/watsonx",
@ -851,6 +853,7 @@ class LiteLLMRoutes(enum.Enum):
"/team/member_add",
"/team/member_delete",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/member_update",
"/team/{team_id}/member/{user_id}/reset_spend",
"/team/permissions_list",
@ -3906,6 +3909,7 @@ class SpendLogsMetadata(TypedDict):
autorouter_savings: ReadOnly[float | None] # stamped by the logging payload; None = not auto-routed
litellm_gateway_injected_cache: ReadOnly[str | None]
router_metadata: ReadOnly[SpendLogsRouterMetadata | None] # None = deployment not flagged internal_router_model
azure_spillover: ReadOnly[AzureSpillover | None] # None = Azure did not report spillover
class SpendLogsPayload(TypedDict):

View file

@ -17,6 +17,7 @@ if TYPE_CHECKING:
AUTO_ROUTER_LICENSE_FEATURE: Final = "auto_router"
LICENSE_ALL_FEATURES: Final = "*"
AUTO_ROUTER_LICENSE_REMEDY: Final = "A LiteLLM license with the 'auto_router' feature lifts the limit."
@ -153,17 +154,21 @@ class LicenseCheck:
return False
return team_count > _max_teams_in_license
def grants_feature(self, feature: str) -> bool:
if self.airgapped_license_data is None:
return False
allowed_features: Final = self.airgapped_license_data.get("allowed_features")
granted: Final = allowed_features if isinstance(allowed_features, list) else (allowed_features,)
return feature in granted or LICENSE_ALL_FEATURES in granted
def auto_router_capability_limit(self) -> int | None:
"""
How many auto-routers may claim each gated classifier or customization capability:
unlimited (None) only when the signed license lists the auto_router
feature, otherwise one per capability. A license verified through the API carries no
feature list, so it does not lift the limit either.
unlimited (None) only when the signed license lists the auto_router feature or the
"*" wildcard that grants every feature, otherwise one per capability. A license verified
through the API carries no feature list, so it does not lift the limit either.
"""
if self.airgapped_license_data is None:
return 1
allowed_features: Final = self.airgapped_license_data.get("allowed_features")
if isinstance(allowed_features, list) and AUTO_ROUTER_LICENSE_FEATURE in allowed_features:
if self.grants_feature(AUTO_ROUTER_LICENSE_FEATURE):
return None
return 1

View file

@ -31,6 +31,7 @@ _PROXY_ADMIN_VIEW_ONLY_BLOCKED_ROUTES: Final = frozenset(
# team
"/team/new",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/update",
"/team/delete",
"/team/block",
@ -767,6 +768,7 @@ class RouteChecks:
"/user/bulk_update",
"/team/new",
"/management/v1/teams/{team_id}/members/bulk_delete",
"/management/v1/teams/{team_id}/members/bulk_update",
"/team/update",
"/team/delete",
"/model/new",

View file

@ -569,15 +569,15 @@ lite --base-url https://your-proxy.example.com configure claude --api-key sk-...
claude
```
The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-<UTF-8 hex of the group name>` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute up` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control
The key comes from `--api-key` (or `lite --api-key` / `LITELLM_PROXY_API_KEY`) and is written into `env.ANTHROPIC_AUTH_TOKEN`; without one the command refuses, since a `lite login` credential expires within a day and keeping it fresh would mean Claude Code running `lite` through `apiKeyHelper` on every credential refresh. The command checks the key against `GET /v1/models`, then patches `~/.claude/settings.json`: `env.ANTHROPIC_BASE_URL`, the credential, and `env.ENABLE_TOOL_SEARCH` and `env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` when those are missing, so Claude Code's `/model` picker lists the proxy's models (under `claude-router-<UTF-8 hex of the group name>` for a group whose id contains neither `claude` nor `anthropic`, since Claude Code lists only those) and you pick between them as usual. Claude Code keeps its own default model until you switch, so that id has to exist on the proxy for the first message to go through; `--model` (or the interactive prompt below) sets the model Claude Code starts on instead, as the top-level `model` key and as `env.ANTHROPIC_MODEL`, both of which have to be on `/v1/models` for the key. The second one matters for `claude -c` and `claude --resume`: a resumed session otherwise re-sends the model its transcript recorded, which behind an auto-router with `return_raw_model_name: true` is the tier model that answered, and a key scoped to the router alias gets a 403 for it; `ANTHROPIC_MODEL` outranks the transcript on resume. Nothing forces Claude Code's sub-agent or background tiers onto a proxy model, so those built-in ids need to exist on the proxy too; `lite autoroute start` is the mode that pins every tier to one group. Claude Code treats a name it does not know as an unknown model: it prints a one-line `unrecognized_model` note, assumes a 200k context window (the proxy appends `[1m]` for a group whose configured or known input window reaches 1M) and sends no thinking parameters for it, so name the group like a Claude model id to change that. The other credential slots (`env.ANTHROPIC_API_KEY`, a stale `env.ANTHROPIC_AUTH_TOKEN` or `apiKeyHelper`) are removed so they cannot fight the one written. Every other setting is preserved and the file is written atomically with owner-only permissions; if `settings.json` is a symlink into a dotfiles repository, the key is written through to that target and the command says so, so keep it out of version control
Plain `lite configure`, with no agent named, asks which agents to wire and which gateway model each starts on, picked from `/v1/models` with a type-to-filter prompt. All choices and selected config files are checked before the first settings write. If a later filesystem write fails, the output identifies each agent already configured and its undo command
What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute up` session holds a backup, and that check comes before any request
What the command changed is recorded in `~/.litellm/claude_configure_state.json` (previous values plus fingerprints of what was written, never a second copy of the key). `lite unconfigure claude` restores each of those keys only if it still holds what `configure` wrote, so anything you changed since is left alone and named in the output; a `settings.json` or `env` object that only existed because of `configure` is removed again. Ownership moves only by a write: running `configure` again (a re-login is one) refreshes the record only for the keys its merge changed, keeps the original snapshot of a key that still holds what it wrote, and snapshots afresh a key you changed in between, so `unconfigure` brings back whatever the repeat displaced and never adopts your edit as its own. A credential (`env.ANTHROPIC_API_KEY`, `env.ANTHROPIC_AUTH_TOKEN`, `apiKeyHelper`) is put back only when the restored file points at the `ANTHROPIC_BASE_URL` it was captured next to; otherwise it stays removed, the output says which server it belonged to, and the receipt is kept so pointing the URL back and running `unconfigure` again finishes the job. It also undoes `lite login --config-claude`, which writes through the same path. Both refuse to run while a `lite up` or `lite autoroute start` session holds a backup, and that check comes before any request
#### Routed model and savings in the status line
`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute up` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline:
`lite configure claude`, `lite login --config-claude`, `lite up` and `lite autoroute start` also install a status line (`~/.litellm/statusline.py`, registered as `statusLine` in `~/.claude/settings.json` unless you already run one) that shows which model the auto-router actually served the last turn and, once the proxy has recorded the session, what the session cost against the router's savings baseline:
```
Routed to: claude-haiku-4-5 -63% vs Claude Opus 5
@ -597,7 +597,7 @@ After upgrading the CLI, rerun your original `lite configure claude` command wit
#### Install the CLI
`lite autoroute up` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
`lite autoroute start` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
```bash
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install.sh | sh
@ -610,7 +610,7 @@ curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch-or-commit>/
LITELLM_CLI_REF=<branch-or-commit> sh
```
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute up`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute start`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
Point the CLI at your real proxy and key before running any `lite model-groups` or `lite autoroute` command -- like every other command in this CLI, they read `LITELLM_PROXY_URL`/`LITELLM_PROXY_API_KEY` (or `--base-url`/`--api-key`), no `lite login` required:
@ -637,44 +637,46 @@ An interactive wizard. It runs the same model-group discovery as above, splits t
The wizard writes the result to `~/.litellm/autorouter/config.yaml` with `0600` permissions, since the file embeds your real proxy API key. Every model referenced anywhere in that config -- tier targets, the classifier model, the embedding model -- becomes its own `litellm_proxy/<model-name>` deployment whose `api_base` and `api_key` point back at your real proxy. That is the trick that keeps your real proxy's config untouched: every actual network call this generates, whether it is the routed completion, an LLM-classifier call, or an embedding call, forwards transparently through your real, already-running proxy with your real key.
You do not need to tell Claude Code to request `autorouter` by name yourself: `lite autoroute up` also sets the top-level `model` and `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`, `ANTHROPIC_DEFAULT_OPUS_MODEL` and `ANTHROPIC_DEFAULT_FABLE_MODEL` to `autorouter` in `~/.claude/settings.json` (and `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when missing, like every other wiring), so every one of Claude Code's own model tiers requests it directly regardless of `/model` or whatever it defaults to otherwise. (A bare `model_name: "*"` deployment looks like the obvious way to catch any request instead, but litellm's Router looks up auto-router deployments by the literal requested model string with no wildcard resolution, so a `"*"` entry would never actually match real traffic -- these env var overrides are what makes it work.)
You do not need to tell Claude Code to request `autorouter` by name yourself: `lite autoroute start` also sets the top-level `model` and `ANTHROPIC_DEFAULT_SONNET_MODEL`, `ANTHROPIC_DEFAULT_HAIKU_MODEL`, `ANTHROPIC_DEFAULT_OPUS_MODEL` and `ANTHROPIC_DEFAULT_FABLE_MODEL` to `autorouter` in `~/.claude/settings.json` (and `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` to `1` when missing, like every other wiring), so every one of Claude Code's own model tiers requests it directly regardless of `/model` or whatever it defaults to otherwise. (A bare `model_name: "*"` deployment looks like the obvious way to catch any request instead, but litellm's Router looks up auto-router deployments by the literal requested model string with no wildcard resolution, so a `"*"` entry would never actually match real traffic -- these env var overrides are what makes it work.)
You must run `configure` at least once before `up`; running `up` first fails with a clear error telling you to configure first.
You must run `configure` at least once before `start`; running `start` first fails with a clear error telling you to configure first.
#### Launch the Ephemeral Auto-Router Proxy
```bash
lite autoroute up
lite autoroute start
```
Starts a local, throwaway litellm proxy on `127.0.0.1:5483` (override with `--port`), running the config `configure` generated, with a self-issued API key baked in (your real proxy key never leaves the generated config -- it only appears there, forwarding to your real proxy). Both the port and the key are stable across runs: the key is minted once, persisted inside the generated config, and reused by every later `up` (and carried forward when you re-run `configure`), so anything you configured against one session keeps working in the next. If the port is already taken, `up` refuses with a clear error instead of silently moving to another one. It waits for the ephemeral proxy to report healthy, then patches `~/.claude/settings.json` the same way `lite up` does, except with a static `ANTHROPIC_AUTH_TOKEN` env var instead of an `apiKeyHelper`, since this key is self-issued rather than something needing SSO refresh. Any `claude` session started afterward, from any terminal, routes through the ephemeral proxy.
Starts a local, throwaway litellm proxy on `127.0.0.1:5483` (override with `--port`), running the config `configure` generated, with a self-issued API key baked in (your real proxy key never leaves the generated config -- it only appears there, forwarding to your real proxy). Both the port and the key are stable across runs: the key is minted once, persisted inside the generated config, and reused by every later `start` (and carried forward when you re-run `configure`), so anything you configured against one session keeps working in the next. If the port is already taken, `start` refuses with a clear error instead of silently moving to another one. It waits for the ephemeral proxy to report healthy, then patches `~/.claude/settings.json` the same way `lite up` does, except with a static `ANTHROPIC_AUTH_TOKEN` env var instead of an `apiKeyHelper`, since this key is self-issued rather than something needing SSO refresh. Any `claude` session started afterward, from any terminal, routes through the ephemeral proxy.
`lite autoroute up` runs in the foreground and streams the ephemeral proxy's own log file into your terminal, so you can watch its routing decisions -- which tier and model got picked for each request -- as you use Claude Code normally. Press Ctrl-C (or send SIGTERM) to stop it; this kills the child proxy process and restores your original Claude Code settings, in that order.
`lite autoroute start` runs in the foreground and streams the ephemeral proxy's own log file into your terminal, so you can watch its routing decisions -- which tier and model got picked for each request -- as you use Claude Code normally. Press Ctrl-C (or send SIGTERM) to stop it; this kills the child proxy process and restores your original Claude Code settings, in that order.
#### Recover From an Unclean Shutdown
```bash
lite autoroute down
lite autoroute stop
```
If the `lite autoroute up` process dies uncleanly -- `kill -9`, a crash -- rather than being stopped with Ctrl-C, `down` is the manual recovery path: it kills any leftover ephemeral proxy process found via a recorded pid file and restores Claude Code's settings from whatever backup is on disk.
If the `lite autoroute start` process dies uncleanly -- `kill -9`, a crash -- rather than being stopped with Ctrl-C, `stop` is the manual recovery path: it kills any leftover ephemeral proxy process found via a recorded pid file and restores Claude Code's settings from whatever backup is on disk.
#### Example
```bash
lite autoroute configure
lite autoroute up
lite autoroute start
# use Claude Code as normal in another terminal; routing decisions stream live
lite autoroute down # only needed if `up` was killed uncleanly instead of Ctrl-C'd
lite autoroute stop # only needed if `start` was killed uncleanly instead of Ctrl-C'd
```
The previous names, `lite autoroute up` and `lite autoroute down`, still work as hidden aliases of `start` and `stop`: each prints a deprecation notice on stderr and will be removed in a future release
#### Caveats
Adaptive mode's learned state does not persist across `lite autoroute up` sessions -- there is no local database, so every session starts adaptive selection cold. A Claude Code session already running before `up` started, or still running when it stops, keeps whatever settings it loaded at its own startup; like `lite up`, this is a one-time file patch and restore, not a live traffic interceptor. Only Claude Code is supported, for the same reason as `lite up`: no other supported agent (for example Cursor) has an equivalent hot-patchable config file.
Adaptive mode's learned state does not persist across `lite autoroute start` sessions -- there is no local database, so every session starts adaptive selection cold. A Claude Code session already running before `start` ran, or still running when it stops, keeps whatever settings it loaded at its own startup; like `lite up`, this is a one-time file patch and restore, not a live traffic interceptor. Only Claude Code is supported, for the same reason as `lite up`: no other supported agent (for example Cursor) has an equivalent hot-patchable config file.
A session that outlives `up` (or is still running the moment you stop it) keeps sending requests, master key included, to that now-freed loopback port until you restart it. Once the ephemeral proxy process exits, nothing stops another local account on the same machine from binding that same port and receiving those requests instead -- and since the port is a fixed, predictable default and the master key is a static value that persists across sessions (unlike `lite up`'s `apiKeyHelper`, which is re-resolved per request), whoever receives them gets a live-looking token along with the prompt content. Restart any Claude Code session before you consider the machine clean, run `lite autoroute down` promptly rather than leaving a stopped session's settings patched, and do not run `lite autoroute up` on a shared or multi-tenant host. To rotate the persisted key, delete the `master_key` line from `~/.litellm/autorouter/config.yaml`; the next `up` mints a fresh one (deleting the whole file works too, but then `configure` must be re-run first).
A session that outlives `start` (or is still running the moment you stop it) keeps sending requests, master key included, to that now-freed loopback port until you restart it. Once the ephemeral proxy process exits, nothing stops another local account on the same machine from binding that same port and receiving those requests instead -- and since the port is a fixed, predictable default and the master key is a static value that persists across sessions (unlike `lite up`'s `apiKeyHelper`, which is re-resolved per request), whoever receives them gets a live-looking token along with the prompt content. Restart any Claude Code session before you consider the machine clean, run `lite autoroute stop` promptly rather than leaving a stopped session's settings patched, and do not run `lite autoroute start` on a shared or multi-tenant host. To rotate the persisted key, delete the `master_key` line from `~/.litellm/autorouter/config.yaml`; the next `start` mints a fresh one (deleting the whole file works too, but then `configure` must be re-run first).
Do not run `lite up` and `lite autoroute up` at the same time. Each patches `~/.claude/settings.json` and keeps its own separate backup, with no coordination between them: whichever one you stop or crash out of last is the one whose backup gets restored, which can silently leave the *other* mode's settings (a static master key and a now-dead loopback URL, or a stale `apiKeyHelper`) active. Run `lite down` or `lite autoroute down` (whichever applies) before switching to the other mode.
Do not run `lite up` and `lite autoroute start` at the same time. Each patches `~/.claude/settings.json` and keeps its own separate backup, with no coordination between them: whichever one you stop or crash out of last is the one whose backup gets restored, which can silently leave the *other* mode's settings (a static master key and a now-dead loopback URL, or a stale `apiKeyHelper`) active. Run `lite down` or `lite autoroute stop` (whichever applies) before switching to the other mode.
## Environment Variables

View file

@ -51,7 +51,7 @@ def _ensure_master_key() -> str:
The generated config is the single home of the key: the proxy server authenticates against
general_settings.master_key only (a key under litellm_settings is silently ignored, which
would leave the ephemeral proxy with no real auth), and the file is written 0600 via
secure_create. Reusing that persisted value keeps the key stable across `up` runs, so a
secure_create. Reusing that persisted value keeps the key stable across `start` runs, so a
client configured against one session keeps working in the next.
"""
with open(CONFIG_PATH, "r") as f:
@ -88,15 +88,18 @@ def configure(ctx: click.Context) -> None:
run_configure_wizard(ctx)
@autoroute_group.command("up")
@click.option(
_PORT_OPTION: Final = click.option(
"--port",
type=click.IntRange(1, 65535),
default=DEFAULT_AUTOROUTE_PORT,
show_default=True,
help="Loopback port for the ephemeral proxy; stable across runs so configured clients keep working.",
)
def up(port: int) -> None:
@autoroute_group.command("start")
@_PORT_OPTION
def start(port: int) -> None:
"""Launch the ephemeral auto-router proxy and route Claude Code through it"""
if not CONFIG_PATH.exists():
raise click.ClickException("No config found. Run `lite autoroute configure` first.")
@ -104,7 +107,7 @@ def up(port: int) -> None:
missing: Final = missing_proxy_runtime_modules()
if missing:
raise click.ClickException(
"lite autoroute up launches a local litellm proxy, which needs the proxy runtime that the "
"lite autoroute start launches a local litellm proxy, which needs the proxy runtime that the "
f"thin `litellm[cli]` install does not include (missing: {', '.join(missing)}). Install the "
"proxy runtime with `uv tool install --force 'litellm[proxy]'`, or to QA a branch, "
"`curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch>/scripts/install.sh | "
@ -117,14 +120,14 @@ def up(port: int) -> None:
raise click.ClickException(str(e))
if existing_pid is not None and is_running(existing_pid.pid):
raise click.ClickException(
"An ephemeral proxy is already running (lite autoroute up looks already active). "
"Run `lite autoroute down` first."
"An ephemeral proxy is already running (lite autoroute start looks already active). "
"Run `lite autoroute stop` first."
)
if AUTOROUTE_BACKUP_PATH.exists():
raise click.ClickException(
f"{AUTOROUTE_BACKUP_PATH} already exists -- `lite autoroute up` looks like it's already "
"running (or crashed without cleanup). Run `lite autoroute down` first."
f"{AUTOROUTE_BACKUP_PATH} already exists -- `lite autoroute start` looks like it's already "
"running (or crashed without cleanup). Run `lite autoroute stop` first."
)
if port == 4000:
@ -135,8 +138,8 @@ def up(port: int) -> None:
if not is_port_available(port):
raise click.ClickException(
f"Port {port} on 127.0.0.1 is already in use. If a previous `lite autoroute up` is still "
"running or crashed, run `lite autoroute down`; otherwise pick a different port with --port."
f"Port {port} on 127.0.0.1 is already in use. If a previous `lite autoroute start` is still "
"running or crashed, run `lite autoroute stop`; otherwise pick a different port with --port."
)
master_key: Final = _ensure_master_key()
@ -196,7 +199,7 @@ def up(port: int) -> None:
click.echo("\nStopped ephemeral proxy and restored Claude Code settings.")
click.echo(
f"Restart any Claude Code session still open from this session, or another local account could "
f"bind the now-free port {port} and receive its requests. Do not use `lite autoroute up` on a "
f"bind the now-free port {port} and receive its requests. Do not use `lite autoroute start` on a "
f"shared or multi-tenant host."
)
@ -214,13 +217,13 @@ def up(port: int) -> None:
_teardown()
@autoroute_group.command("down")
def down() -> None:
@autoroute_group.command("stop")
def stop() -> None:
"""Restore Claude Code settings and stop a leftover ephemeral proxy, if any"""
try:
record: PidRecord | None = read_pid_record()
except ClaudeSettingsError as e:
# down is the crash-recovery path -- a corrupt pid record must not block it; clear the
# stop is the crash-recovery path -- a corrupt pid record must not block it; clear the
# unusable record and keep going rather than leaving the user with no way to clean up.
click.echo(f"{e} Clearing it and continuing cleanup.", err=True)
record = None
@ -238,7 +241,34 @@ def down() -> None:
elif restored.existed:
click.echo(f"Restored {CLAUDE_SETTINGS_PATH} to its original contents.")
else:
click.echo(f"Removed {CLAUDE_SETTINGS_PATH} (it did not exist before `lite autoroute up`).")
click.echo(f"Removed {CLAUDE_SETTINGS_PATH} (it did not exist before `lite autoroute start`).")
AUTOROUTE_ALIAS_DEPRECATION_NOTICE: Final = (
"`lite autoroute {retired}` is deprecated and will be removed in a future release; "
"run `lite autoroute {current}` instead, it takes the same options."
)
def _warn_deprecated_alias(retired: str, current: str) -> None:
click.secho(AUTOROUTE_ALIAS_DEPRECATION_NOTICE.format(retired=retired, current=current), err=True, fg="yellow")
@autoroute_group.command("up", hidden=True)
@_PORT_OPTION
@click.pass_context
def up(ctx: click.Context, port: int) -> None:
"""Deprecated alias of `lite autoroute start`"""
_warn_deprecated_alias("up", "start")
ctx.invoke(start, port=port)
@autoroute_group.command("down", hidden=True)
@click.pass_context
def down(ctx: click.Context) -> None:
"""Deprecated alias of `lite autoroute stop`"""
_warn_deprecated_alias("down", "stop")
ctx.invoke(stop)
__all__ = ["autoroute_group"]

View file

@ -214,7 +214,7 @@ def build_generated_proxy_config(config: AutorouteConfig, master_key: str) -> di
def master_key_from_config(config: dict[str, JsonValue]) -> str | None:
"""The master key persisted in a generated config, or None when absent or blank.
Single definition of "this config already has a usable key", shared by `up` (reuse
Single definition of "this config already has a usable key", shared by `start` (reuse
instead of minting) and the configure wizard (carry the key forward on rewrite) so the
two sites can never disagree on what counts as one. Returned verbatim, never stripped:
the proxy authenticates against the exact bytes under general_settings.master_key, so a

Some files were not shown because too many files have changed in this diff Show more