From ccd8c4112c2092d680fcfc60a2b20da1b722d5e9 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 17:38:48 -0700 Subject: [PATCH 01/30] docs(readme): add Deploy on AWS/GCP with Terraform section Adds a quickstart for the two published Terraform modules on the public registry (BerriAI/litellm/aws and BerriAI/litellm/google). Copy-paste main.tf for each cloud, the one-time GCP Artifact Registry remote-repo command, and pointers to the registry pages for the full input surface. Sits inside the Get Started section, between the gateway/SDK table and Run in Developer Mode -- where someone scanning the README for "how do I deploy this" will land. Co-Authored-By: Claude Opus 4.7 --- README.md | 118 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 118 insertions(+) diff --git a/README.md b/README.md index d600f3952c6..faf79c3729d 100644 --- a/README.md +++ b/README.md @@ -404,6 +404,124 @@ You can use LiteLLM through either the Proxy Server or Python SDK. Both give you Support for more providers. Missing a provider or LLM Platform, raise a [feature request](https://github.com/BerriAI/litellm/issues/new?assignees=&labels=enhancement&projects=&template=feature_request.yml&title=%5BFeature%5D%3A+). +### Deploy on AWS or GCP with Terraform + +Run the LiteLLM proxy as a production-ready componentized stack (gateway, backend, UI on separate services; managed Postgres + Redis + object store) using the published Terraform modules. Both modules are on the [public Terraform Registry](https://registry.terraform.io/namespaces/BerriAI) — no auth needed. + +#### AWS — ECS Fargate + Aurora + ElastiCache + ALB + +[Module page →](https://registry.terraform.io/modules/BerriAI/litellm/aws/latest) + +```hcl +# main.tf +terraform { + required_version = ">= 1.6.0" + required_providers { + aws = { source = "hashicorp/aws", version = "~> 5.60" } + } +} + +provider "aws" { + region = "us-west-2" +} + +module "litellm" { + source = "BerriAI/litellm/aws" + version = "~> 1.89" + + region = "us-west-2" + azs = ["us-west-2a", "us-west-2b"] + tenant = "acme" + env = "prod" + + # Production: provide an ACM cert. Without one, set allow_plaintext_alb = true + # (dev/trial only). + # acm_certificate_arn = "arn:aws:acm:us-west-2:111122223333:certificate/..." + allow_plaintext_alb = true +} + +output "litellm_url" { + value = module.litellm.alb_dns_name +} +``` + +```bash +terraform init +terraform apply +``` + +Provider API keys live in AWS Secrets Manager; reference ARNs via `gateway_extra_secrets`. Full input list and architecture diagram on the [registry page](https://registry.terraform.io/modules/BerriAI/litellm/aws/latest?tab=inputs). + +#### GCP — Cloud Run + Cloud SQL + Memorystore + HTTPS LB + +[Module page →](https://registry.terraform.io/modules/BerriAI/litellm/google/latest) + +Cloud Run can't pull from `ghcr.io` directly, so first set up a one-time Artifact Registry remote repo backed by GHCR: + +```bash +gcloud artifacts repositories create litellm \ + --location=us-central1 \ + --repository-format=docker \ + --mode=remote-repository \ + --remote-docker-repo=https://ghcr.io \ + --project=my-gcp-project +``` + +Then: + +```hcl +# main.tf +terraform { + required_version = ">= 1.6.0" + required_providers { + google = { source = "hashicorp/google", version = "~> 6.10" } + google-beta = { source = "hashicorp/google-beta", version = "~> 6.10" } + } +} + +provider "google" { project = "my-gcp-project"; region = "us-central1" } +provider "google-beta" { project = "my-gcp-project"; region = "us-central1" } + +module "litellm" { + source = "BerriAI/litellm/google" + version = "~> 1.89" + + project_id = "my-gcp-project" + region = "us-central1" + tenant = "acme" + env = "prod" + + image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai" + + # Production: provide DNS already pointing at the LB IP for Google-managed certs. + # Without one, set allow_plaintext_lb = true (dev/trial only). + # lb_domains = ["proxy.example.com"] + allow_plaintext_lb = true +} + +output "litellm_url" { + value = module.litellm.load_balancer_url +} +``` + +```bash +terraform init +terraform apply +``` + +Provider API keys live in Secret Manager; reference resource IDs (e.g. `projects/my-gcp-project/secrets/openai-api-key`) via `gateway_extra_secrets`. Full input list and architecture diagram on the [registry page](https://registry.terraform.io/modules/BerriAI/litellm/google/latest?tab=inputs). + +#### Both stacks include + +- The full componentized split (gateway / backend / UI as independent services) +- Managed Postgres (writer + reader) and Redis +- Versioned object store for proxy state + file uploads +- An auto-generated `LITELLM_MASTER_KEY` in your cloud's secret manager +- A one-off migration job that runs `prisma migrate deploy` before the proxy starts +- The same `proxy_config` surface as the [Helm chart](./helm/litellm/) — pass YAML as a typed map + +The Terraform modules live at [`terraform/litellm/aws/`](./terraform/litellm/aws/) and [`terraform/litellm/gcp/`](./terraform/litellm/gcp/) in this repo; the registry entries are read-only mirrors updated on each release. + ### Run in Developer Mode #### Services 1. Setup .env file in root From 260f5f1371e3308810cdaab61daa4da231cc1e76 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 17:44:49 -0700 Subject: [PATCH 02/30] docs(readme): add 1-click deploy buttons for AWS + GCP GCP gets the real 1-click: Open in Cloud Shell badge that clones the repo and walks through `terraform apply` via the existing DeployStack tutorial (already shipped at terraform/litellm/gcp/examples/default/ TUTORIAL.md). User just picks a project. AWS gets a soft 1-click: a Launch in AWS CloudShell badge that opens an in-browser, already-authenticated shell. User runs four commands (clone + cd + cp tfvars + terraform apply) once inside. There's no native AWS deeplink that pre-clones a repo + runs a tutorial -- CFN "Launch Stack" + CodeBuild would be needed for that, and that's a separate piece of work. Co-Authored-By: Claude Opus 4.7 --- README.md | 17 ++++++++++++++++- 1 file changed, 16 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index faf79c3729d..d304d15f169 100644 --- a/README.md +++ b/README.md @@ -410,8 +410,19 @@ Run the LiteLLM proxy as a production-ready componentized stack (gateway, backen #### AWS — ECS Fargate + Aurora + ElastiCache + ALB +[![Launch in AWS CloudShell](https://img.shields.io/badge/Launch-AWS_CloudShell-FF9900?logo=amazon-aws&logoColor=white)](https://console.aws.amazon.com/cloudshell/home) — opens an in-browser shell, already authenticated to your AWS account. Once inside, run: + +```bash +git clone https://github.com/BerriAI/litellm.git +cd litellm/terraform/litellm/aws/examples/default +cp terraform.tfvars.example terraform.tfvars # edit region/tenant/env +terraform init && terraform apply +``` + [Module page →](https://registry.terraform.io/modules/BerriAI/litellm/aws/latest) +Or call the module from your own root config: + ```hcl # main.tf terraform { @@ -454,9 +465,13 @@ Provider API keys live in AWS Secrets Manager; reference ARNs via `gateway_extra #### GCP — Cloud Run + Cloud SQL + Memorystore + HTTPS LB +[![Open in Cloud Shell](https://gstatic.com/cloudssh/images/open-btn.svg)](https://ssh.cloud.google.com/cloudshell/editor?cloudshell_git_repo=https%3A%2F%2Fgithub.com%2FBerriAI%2Flitellm&cloudshell_workspace=terraform%2Flitellm%2Fgcp%2Fexamples%2Fdefault&cloudshell_tutorial=TUTORIAL.md&cloudshell_image=gcr.io/ds-artifacts-cloudshell/deploystack_custom_image&shellonly=true) + +Real 1-click. Opens Cloud Shell, clones this repo, and walks you through `terraform apply` via a built-in [DeployStack tutorial](./terraform/litellm/gcp/examples/default/TUTORIAL.md) — pick the project, the tutorial sets up the Artifact Registry remote repo, writes `terraform.tfvars` from your answers, and runs apply. + [Module page →](https://registry.terraform.io/modules/BerriAI/litellm/google/latest) -Cloud Run can't pull from `ghcr.io` directly, so first set up a one-time Artifact Registry remote repo backed by GHCR: +To call the module from your own config instead, Cloud Run can't pull from `ghcr.io` directly, so first set up a one-time Artifact Registry remote repo backed by GHCR: ```bash gcloud artifacts repositories create litellm \ From a505734a2cddcece2d27ae7dea2e192f0f8b7a13 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 17:50:53 -0700 Subject: [PATCH 03/30] docs(readme): move AWS + GCP deploy buttons next to Render button --- README.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/README.md b/README.md index d304d15f169..7802e33baee 100644 --- a/README.md +++ b/README.md @@ -10,6 +10,8 @@ Deploy on Railway + Launch in AWS CloudShell + Open in Cloud Shell

LiteLLM Proxy Server (AI Gateway) | Hosted Proxy | Enterprise Tier | Website

From d0ff1753973ee16b2c9ef88d34d2647e3d7cbec4 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 17:54:38 -0700 Subject: [PATCH 04/30] docs(readme): unify deploy button sizes and badge styles --- README.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 7802e33baee..2509b761563 100644 --- a/README.md +++ b/README.md @@ -6,12 +6,12 @@

Open Source AI Gateway for 100+ LLMs. Self-hosted. Enterprise-ready. Call any LLM in OpenAI format.

- Deploy to Render + Deploy to Render - Deploy on Railway + Deploy on Railway - Launch in AWS CloudShell - Open in Cloud Shell + Deploy on AWS + Deploy on GCP

LiteLLM Proxy Server (AI Gateway) | Hosted Proxy | Enterprise Tier | Website

From daba177a5205f1be70a658a5dd063b23fd67e21f Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 18:15:36 -0700 Subject: [PATCH 05/30] docs(readme): bump deploy button height to 48 to match Render/Railway --- README.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 2509b761563..693aee14683 100644 --- a/README.md +++ b/README.md @@ -6,12 +6,12 @@

Open Source AI Gateway for 100+ LLMs. Self-hosted. Enterprise-ready. Call any LLM in OpenAI format.

- Deploy to Render + Deploy to Render - Deploy on Railway + Deploy on Railway - Deploy on AWS - Deploy on GCP + Deploy on AWS + Deploy on GCP

LiteLLM Proxy Server (AI Gateway) | Hosted Proxy | Enterprise Tier | Website

From 1a18184f0eae4b331946ae564fd51dc7dbf1a0a2 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 18:25:49 -0700 Subject: [PATCH 06/30] docs(readme): bump AWS/GCP badge height to compensate for SVG padding --- README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 693aee14683..d97f07a1b8b 100644 --- a/README.md +++ b/README.md @@ -10,8 +10,8 @@ Deploy on Railway - Deploy on AWS - Deploy on GCP + Deploy on AWS + Deploy on GCP

LiteLLM Proxy Server (AI Gateway) | Hosted Proxy | Enterprise Tier | Website

From 8ba452c0ecfaf6b81268eb596ba2fb0d463bee06 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 18:26:21 -0700 Subject: [PATCH 07/30] docs(readme): bump AWS/GCP badge height to 72 --- README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index d97f07a1b8b..91246629ad6 100644 --- a/README.md +++ b/README.md @@ -10,8 +10,8 @@ Deploy on Railway - Deploy on AWS - Deploy on GCP + Deploy on AWS + Deploy on GCP

LiteLLM Proxy Server (AI Gateway) | Hosted Proxy | Enterprise Tier | Website

From 2c56b4d3f64ae941322b78b35df4bf0dfea51f34 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 18:27:10 -0700 Subject: [PATCH 08/30] docs(readme): bump AWS/GCP badge height to 84 --- README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 91246629ad6..07a74296fd1 100644 --- a/README.md +++ b/README.md @@ -10,8 +10,8 @@ Deploy on Railway - Deploy on AWS - Deploy on GCP + Deploy on AWS + Deploy on GCP

LiteLLM Proxy Server (AI Gateway) | Hosted Proxy | Enterprise Tier | Website

From 65890b732e1e67af0d3c09a318197534d8966ced Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 7 Jun 2026 01:33:43 +0000 Subject: [PATCH 09/30] fix(readme): make deploy buttons same height (48px) https://claude.ai/code/session_01MxQRMHSDXbqJh74rF86UBc --- README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 07a74296fd1..693aee14683 100644 --- a/README.md +++ b/README.md @@ -10,8 +10,8 @@ Deploy on Railway - Deploy on AWS - Deploy on GCP + Deploy on AWS + Deploy on GCP

LiteLLM Proxy Server (AI Gateway) | Hosted Proxy | Enterprise Tier | Website

From 97f067bc70d7006f7aeb8dc117412c4d0f67e514 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Sat, 6 Jun 2026 18:37:28 -0700 Subject: [PATCH 10/30] docs(readme): flag GCP project ID substitution in image_registry --- README.md | 1 + 1 file changed, 1 insertion(+) diff --git a/README.md b/README.md index 693aee14683..529e1de3354 100644 --- a/README.md +++ b/README.md @@ -508,6 +508,7 @@ module "litellm" { tenant = "acme" env = "prod" + # Replace my-gcp-project with your GCP project ID (same value as project_id above). image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai" # Production: provide DNS already pointing at the LB IP for Google-managed certs. From 99e1a7612860fe469af44a5ef08434867359fd41 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 7 Jun 2026 02:10:22 +0000 Subject: [PATCH 11/30] docs(readme): equalize deploy button heights and fix Cloud Shell button font GitHub rewrites an image's height attribute to "height: auto; max-height: Npx", which only caps and never stretches, so each image renders at its intrinsic height. The AWS/GCP shields badges are intrinsically 28px while the Render/Railway buttons are 40px, leaving the row uneven regardless of the height="48" we set. Replace the two shields badges with committed 40px PNGs so all four header buttons render at the same 40px. Also swap the Cloud Shell button from open-btn.svg to open-btn.png. The SVG renders its label as live text with font-family "Roboto, Sans" and no generic fallback; since neither font exists in GitHub's render environment, the text fell back to a serif (Times New Roman). The PNG bakes in the correct typeface. --- .github/deploy-on-aws.png | Bin 0 -> 4113 bytes .github/deploy-on-gcp.png | Bin 0 -> 4922 bytes README.md | 10 +++++----- 3 files changed, 5 insertions(+), 5 deletions(-) create mode 100644 .github/deploy-on-aws.png create mode 100644 .github/deploy-on-gcp.png diff --git a/.github/deploy-on-aws.png b/.github/deploy-on-aws.png new file mode 100644 index 0000000000000000000000000000000000000000..06d41f2a5e06cd60aca118d53ea50f0beb234ce8 GIT binary patch literal 4113 zcmc)N)msyQy9aR4fiwf8hor>lPDgi1cS?-zZV)D&0wN;yqoh+A-5^Y20|!!~B9msL zk>h#JA8_u@xp|-O#rNX9df(@h`rJU1oP>!44-b!ATT9It4-Y^7Uw%&X;NN_|FByb~ zM>(ObrfeEsy8kY$fEL9vIS3bM#rAf(dhqKUl{=gEzT{xz0c&JDbyg^RA|A z>#0{c@dF@A;t4P<;5Jr^d|kPC(;i&d-MJopMA7#qYIAn2XMZj1J#2HgPtolV?pOTo z(^>yuD6HVNhm9!FG1-MEkyM2d7^DW&lgEk0P)T22RXMD6j5YVQ!!0YJT^zXB;tdTZ8* zD4<49vO2_+ndh7%j55ppVMME1G(lY;cVu6S^v@lvQ7K}$9#pFE$^_=`oE9UcrOv$O^Mlrb^^4B4RF*)+t?cj`jDyI!nJnnat_Z7h1kWFl&f}(kNUp~O8dRgU+ zO6w@AjOdeH3&L z5)BbCU{O2ad!451r>dcd#AKY}I$Hvz*rFg(*^`B-Ml20S?wFErtCL6GZyZB)k1`R$ zz3{#mabK0$y>p5NsU>RzmI?Y8k}ETK?L4#<4v!Hlgt(u3MDklhwhmA2qeGj}wR*JJ zw>4dQhp_w`Dz&=-=!Hx}kA`Q-GNLWm8^*K-GVuCa^Y^6;wF5rbjH0Vo4V7ekyk*H;9%y#v?mkN4Cx8rVD|c(VAcX&=xbF~r_}VEbT?{f{9-Z9SBJKb zv3?fslP_|o5ocv9O?b!6SzoXgc(7KCT&i`i&7HceUv9tL`&wss_h&waYzB3DU$_iZ z)Q5d|(S|EM!am+y#E ziJY)RwJy}c<;#*aCS>K>f0e?R6>RW6dtKCfsLixK_u2(J-Ywi3C;MHiav~lqt-P`_ ze-Tt5lu%&;nn7vQ?ut4R=IL0%(b2+4#NOp!2WwLO4!1!GI0~9A zl?@e)DssF!GKyxh&xX)EKuwbBrzUh5H9buuNsiZWzV!$#%D&F2Q560GlXr*0QP$9u zTGdN{x%BJE_nw z<*G&HB>dR*ZwSoPnxQp4Hr$Y)i3|a<@b%d6=y*5b zy{TOTqa|l4GZJzs{q%$G{+#fR^o`_R$dmQML++Dz76-TYJpxq(JzH&h1C=e?3g_G4pCIt8O}h2Ia#1?O>Q9RYVymRCHWOLHF+ z2!rt+HqniBVHp}Rc)CrF0FCa$7vdV^?P{0DP=1c9)?oW{>1 z=Jab1i1dm2^2(HA)h}8Dkz3u*DHh{8!o6Dn)5wJhvL* zXgCqA__uXA3#T2Ku%hPYGF>h6p&s!;bl7$;y6Elca5R>xP$6A)Goqv}K4w1r6-n&` z%i04q+=&LOY)EGxr4Osgr#xV=DARhp;7*l-OXv11blSvTRI#+D<+5XKc8Fwh4k1hx zbSc1iCaXEyo_i>?^PJyd;?PfIi(fy9f%25)VutFb=U0RzQzEyb$%|YeLpVZDdE6&Qq(~)dDT*TK$Ydjc2>eR=ioswRaII`1{XT__;Vfx22 z>1v967<=)3njZ-nG1J3U8FSarzWG&&Kv-(Lv@bP3CPdss%Ij}4 zh|Xc)F8u}|hol6uQ8cz3p{##Tl9&0Fobdl;sN&LUXT*Iw<)diA9l@Ohvb>kAM;fp>Jf-ET;a zb!Y_-ffGN9Sb%sc$|`8xqwIo{w<3lnm4p>NhNw^!3x?&uPxHr390O30qG;Wzd*3FZ zcr-kJ_T8OP$fERVjEhfSiqYAhpD^)RelKa50Ie&4E&L*Cu6Y)${HdSX$|OVQm@ zkVv1AwArVHGGOfq*;)o<=k*|KSyDP@Dp={!ptudlP{!9k%}e6rmN>xYiRvI9#Dgy@ zyLePDXZWO1>n?~A3eVM#zoOi^6uNII&QE)NrVg%tD`-#owA8jg{%Kbvl2hD}Mt|ZH zSW-FFxQ!hXnIT`3BlQ@$d{|tw&X-chSeF&=)bY}Jw(FT2g{rzl?1d1d$L3S2PEM}1 zMFV&YXv&+im}rS($~rpts1=@Qi}}gVP?r4ff~G2Q1`zNn5L$IJeJAEggV2ybb^^vf zLy|o1;4{ZrvKznQqtmLn*Ub3i$M2&m(26|eF5GSi3;c{JhSIE$KfYtBVifsAOmhod z&TXZwX_Mtegx2^pG;D!L89yhU4q(OU;9O+cn3f_VgA<#zgn_G-R_brtBVOL%a~(5{ zgjzR1{O|gok7^gqn6N@|y9R8+auKxV+h&YtYme$Cg~heQ$#(brP2* zXA5UCIdRerZU~U7OLvTKO7phMfV~nKaS^j$^RrU_(F0R8w4^1r06w?>*2oNKL$>L; zbpOrA56nVHtY!yF8W{uHBl%MQ2YbF;5;p3A>k8u?E!jPi z>)x07MgnDyh3i^IO_7olx;CRM&1txQ`+9o*UMW?G|6S3r_$WYSMZ!>SFdQM6COfNY zcK$KkJ@ZFP3T6sZp7Px6;`!i@vB(Yv8C@@JzLo>AR%$LKuBZWSB|Yy5-6$0! zPl~~q&j|H*vZ5<77(x+%Bb0jZoMj5B@tFHT!kWCC}*2>>Kr z2cZ$xyxk6-EUpY=Tm0ofo%79u>KEZ*NwV5ja8}7iKWPbXRcTbFH5Y7_d9Ii$sYU(i z4fE?m;@7*KkW^GpZ4!@}c48z2am)h=4RM zEU{HAKZY=*!h_$QX*P*6bSv9z+SIdu*ZMvlU$R+6$IsXp^N^kxq-TTeLUvyDyZW&+ z`1uA_EsqH3WqcS)AC8RiVR{cTd5l=v_l;596rg{VH~!_LnbFg>+h=Dr9IS;>j+cv@ zK)9QStF4}Q@~IaB+ysO_Gbj{gAn7vysyH&@QF5jk98=xKs!6RH4>fkfnxr=o&W)n{ zO6oxyWD{S6o)l(E#8j5;6bMCW|4C#|pPSv9DI%p*nIzYk$(T2LtBIV$HH%f;PXYY< zeXV=uXdbMXX{1O65+}!kNCXe~FfHL7=X-A^bCx{EUWXx+V$0Bz39+^=zg^b*C=T8) z&%9C*X>bi^zS+TeehnSI*5j>ySkT~XxrP5)nYo>+Iufh+q^FCH)d8F8MlB7n5dmWTv1S`V~^p^yps`DjgqhQJ2^8ej?xMn9{kjmn9UKJAcqM7? zTeOSSiiG4unxV!g7I%>!z2_age)xMl|1zjZ6D=B%pyBQ0U*Ky*AH&Yv&rqU}g4!8^ z{wakAG%`%jaK1ghTf;QriNdUMUA~M4f;Ib7MWOcIcRA(@h|(81cmb=&&(@AVirF>! z^~WYzrh94E_wf94 z7riyUDOdRF8LM)%HXeQ@PveH+fPu&D1SRr^q7TtTwbMBZCKF9BPWpaqlEpM|Qo3E$ zXB7@`*r>K|V~Y5cVkea~`lzU$*WOp@DLrd<^56tau(h_;UqhmF$~>LYnY4;$;scX( zMmjQXok8%#uu52bi(Iphdz)~erSY=Y8f|3T#@JK$CW9uz_xMkxEG%Y&c?rgG$n|rO zfPGuroOm5vR3~XZ?TU&NssXqE8au*Cw^ds6|9opY`+PeHH%RGi0nyTKgi=L2Rc0=X z>gXeNiWX^Kqo&pHs}NN7t_~yr1JtW*VuJrKmj3TL-nswh`TKh^iR+Sof$+aY{}2yI XH7&3dhx>acd4x^h`m+psv0$-R&i6K zu}5sGXdd^U@Vt1QH|ILnS=Twg>-XY&eG-fewCL{KyF)@kLZ_px4k95TZMgA`s3~u3 zX#4?ygoKeqM_t7fp0}HSJDp{^v!D1=VN&6Fq>M796DI}B3qHPk>_-0-GFiVZBG1H> zX`vtfv6W2Ncw~rA639~5M;hNnzEmH7NDv~Ae#-%oZomSpAXH9qb8VP*-|l;YCw@y6 zHH592zKigUsg*qY(#k(SsxPl+tRD3Wa%tZI#4!GP?!5FvF)|`rl^PK9o8u1WP0UUW zz<-nR5Q4-os)o|VA{kZVYd@lxRH+YNut?k#b|F+XsR3VSj0XQ}y7zzCRH_cdt;A8T z{w@G)&B+Q)Z*R`Y@@e_q)(z8x1nPqn=Rc$~NcTC7=V-pF7r)tE^6=MKHW_< zx>wFd^Uk={~e5?&eTCKVSyGe7%Ks0f${$W_-y6+C-U$!$NH7Z!lO_?dCmJ zX^<|!2I$GtG~VPyZGZ}v2{3fqbM7v7h? zr<%setjtq^vZlv2pq{|F3Z64jcYt-f0)TjBVeO+5n9^Mv|4n$GO^i2q+ZezZ;F)l; z!z9u1$1i|OduT`$%aAPLtAYO75E|Q~g;@Q!HuiaVm-hAbJKt*PYS5D>Fv91Nkh7t7 z@zc(vTr#l}nPT7k#PmFGb>}k3Vufaqb62>vTzPu$Sbuu+?^o9P!^CWBK?73b?t@|w zzKbUetWEzIdO^JHuh2d<>uGWv+K0e)G>VB0ORvhvd|)>1O&Od*bv>G}$ggzN7$y$L zk*HYJi;F=cxH6q~8b1tpE||C(nRv|eP&=tg5%em?yhL(j*zeB7?!M%{?Xj=okpa`-cpOD?=xedvYS!#OV_8@J_L71sJ5~K#vtAb(X}waa zjbYP)Tj}_~3%G}BSZY>>({aHjuHG-q{sF$u50Kv31-TBw+F{P=IH}gBp4z(Av8)Cv zB@9oyGi}bG*EY@+fSw{gyk0+Q8HHZoi*QzdG9l@GOeh}tTL|~Xg7GN&Cn*2pdJD5z zMonrbIi_Y%2~q!x3T8Do`Xyvx%YHbB{&1_P;04z++*4lAlgo6R?U|9j!PpK8ttRoe zD&DVFQi)x?*>I*1hBW`x22wDT{`<{xYP4S=KWdsPVx#feKFCc+&o)?4z?fm?dusuI z`{8c9pV08%kVYlX^f+hDwf1di8^wBav9LOGTz9ReS2jqn_TRdp?g+21aE=aDMKXp; z&jrqsqh+(T{vy0fiY!nl5Siktz$Vz^a44T)IQj{Wi8dOu8B&vJO>Lf7dR}eojuwqT zP%vodI0qh3Z8#@#uG#>1Qm}10#6KeT?)r{t?blx~=GvlCTgDzqa!u55dNosKc>XB)(5-sl7MdV&3e39Ai{z||i`Ra{ zWspV(0(|r6+7{*KXSJ|6tH)R}jJ$mTYdxk2G2gtG5qPmO>3{Tnnib)?h20dkw_41# z`n;2ePMI(=nGE#LjO3aTEk5l+=Jm7;t;wg(IMZ2e3MjB)^;J~U(b6kOulrjsqR6`Cn$4xS|$Si>53GsfhqfI zclnyjIM(-DUoJs4SLai6;`|HjxGsSkQ%6w1FKD%Ax}l=SJr!`bH0@Yk2N=tf&7I+} zG9@z$C{|>a9&#p+a z8mJ3_sF3G^>EbaRR>zI>c+!^*B{3BAy-Z>weWALT~`xn++pU-e(kQMg7on1@%y>@NS@?QhH+We=I6~1APR`moJfK4^Y+K$}TJR2AB_(;LjkNiX#mf zxpWE%U>&+mZa+;$61STvd%dD^tsGlvaxD3b9HrO2l_2+qgN{dH{pf0X9iPC6@0~H{ zpfP&R_}gkM^-8b>HW?q^WSH9~b6!jQfwD+T<<8@wl2<+nvFL;R#1PClmlct6#!p4M zv&Z=+Gq-2ZqV2}tS)>c9b)f1QGO=%h44pNvCyyNv@Zt#abux9=zWy?xf>H(Y#? ztMKVQuxItlc`H7s(?#x2aUGUBe%#=}yQB$Dron47c`gGvt=I$A92(;A+pq%_J5kCU1$nt{qdb*9u}g9#rd`3FCWq>BlbJ8rrmg&gxgv4TwQY9o;WnA? z<@IGc0*V|=GuuC0*rLnhOQK;pK;?3dpS9@M0KCAMF|m;= zTWtXb!rJJ!lHT2r%$PjZhN$Um$4`zC5*}`sf4FsCl^x2Yv>ZmFbxkbF;~9VGySNx= zr^ARabC{Z^7c!b)fWG(KY{bBLJZs9x^ZQDMZf1)HaiRFZLMkBa)KLmn8nR;6xFNYZ z{$`!KyUJ=T38G(MNxDzl$z@&zs;M-Dj{#Fv#R?AB$Xbys8gj;{VW#Sra;iT9N?!IHQfIf~H^*WBnpVIeM zR4?ic%jEFOThZYT!mE{>TIpQqe|Wki`ONW~sj`_LC`>NY2jBV>0Y^0pB!z99;ewwB zY*w%}e*9pSbMOhJq~aJLt^eu}k&jE{s9F9VkUbM6;3vOZIZ5O>(w_^k26v;&wfU?l zW|0r#SfXQC$bg69w z+OMdP>~`qxY^1Fm_n^Lhj#((#a5L=L1vLFVx=i>qnSC%r)Dpiy;3k==rTI9WMSZ37 zgjUw4@WNWltt$iNFEqx5a)AG87dEFsAVOQ5iubW@_@1~X+VTEP;0A!)zQ!`tl3a;6`$6_u?hZ|dqOXr!NLe!CW5!B z0QSzSrvI`O=`u55`%BN3ce{g|RptYyxIk4fKBMI`nN4`8!PKlkIj2YyKNo@c6}zcf zI^_H~Om6P4?)g=~7HX+Ak>X!E0o;l6mV&n@0wO%2yrjL&270$4wfEPXn6aj*M5eym zs&O_-`Y;vt`j9v-bX=;5{VtR>yHPcs$sRVPRxMm2O>-#u4$AvVTv|Yb=UDcU!)Ctk zJtS4m*6`#l@&L6aGwzeKr2?)mD%9ddhIgkE%lZ89rZJysZiGn3xZ8fe=@}_)O_f=D?j*jXKWQ%u%O-m=X7#`!O{BmwZJ`2j{iY z+QpK5PepKJm{{Sfr1H}VV*eeJm&{>$Li6Wx{QA1N^|xzT`w0A#`P@*#Vk>cZ3c6R? z6YD%#6En{IPIswStsm?p?6+C_I774i6v5rP#&9=>S=Ppx5DU}ZC~racTSM9IvZNxp zHa%C#gXT{<*!yqO;owowC+7yxXY`Q>;s+$ITAM<1)FQY=()HQj&HY@n&z&#)`*;ot zq)S)JTHU-03X7+oSj)U|+krmxqW=S{;i!nf3$L$HRh@6v#{dy-$ZUDydXt*xF_La z1?7Yx!|C?9Vh~sEFO=jAY%*UzL2Ru1T^9{VYnxE2rNe0-Eq||LLK#zr-tO3t8f5s^ zEINl5q_8d~YGI-a-hV)fO$M7o%lZ)eaYNiEI$OXLU&>ccvj$JM3B5b}A(Au&6w>`Y zxB%Sv`4=vV+GbaCV+{6n}1QUMgx87pb4UVN*LW{CbAmM&`b}$U=lh z5F#^$2Wy2(80^cbj{MGb)#AhxJ`hV`%S{y@Z&qSguM+-yx+HsJpz&hRpS9mhB-Xlh z|Ru&Acapd_PSpRn=xr`RHQh(AHtn6=8 zU@U8PvRD1{J&KTjt3QdwzrAN!aY3nfNvz&tV5RSoc<=Errl*aOPcj9zg`VRJy2`e9 z)Ov4d8t#!VLAhO|^O%7OIT~@i@+PAe!4%oTTMVjDB!uwu_y4eyH}&$$%o&@NtF&_Z z4(|pZm(tIl{b})OWt3S@+iRG0!&lh%?JdX6D>T(!kk2;^P|c07pbGmIlU@SbF6!+* z=c%i8)JRj(P@k!M&COZi-eH4|xUxLyY#FyZ7UZr?AyX~z(&SoZHOR7rni94m0})_y z^Gyj0{BKEsYuVDFKeC|s6Xq>ZZoRd-bIwD|^mQRuE95L~zLH^e7j#9EC>nEV-K*EuA= zUIQdnXoIg|-|=<7eJEke9XIn{x~+{y@qxySlDS8d?kX8F+G%I+S~L;TrBs58k6g_m zl)Z7C=gD^bsNolD_u4p!JPDsVV@YEm{yF-WW^?vjQ#9C~iuB}s*8*S&juvvU#r)Gl zVk@O3YAy9^3_{zA_MXA+~u-6(lve$4kwGi^MFX1IT5p`ZDNJLbrH{#*dkc$BBK zcrOq_#qsp{<)_lGHE2~te!XBS6caXa$(*aUO0-SX{E)eq{$MVz

Open Source AI Gateway for 100+ LLMs. Self-hosted. Enterprise-ready. Call any LLM in OpenAI format.

- Deploy to Render + Deploy to Render - Deploy on Railway + Deploy on Railway - Deploy on AWS - Deploy on GCP + Deploy on AWS + Deploy on GCP

LiteLLM Proxy Server (AI Gateway) | Hosted Proxy | Enterprise Tier | Website

@@ -467,7 +467,7 @@ Provider API keys live in AWS Secrets Manager; reference ARNs via `gateway_extra #### GCP — Cloud Run + Cloud SQL + Memorystore + HTTPS LB -[![Open in Cloud Shell](https://gstatic.com/cloudssh/images/open-btn.svg)](https://ssh.cloud.google.com/cloudshell/editor?cloudshell_git_repo=https%3A%2F%2Fgithub.com%2FBerriAI%2Flitellm&cloudshell_workspace=terraform%2Flitellm%2Fgcp%2Fexamples%2Fdefault&cloudshell_tutorial=TUTORIAL.md&cloudshell_image=gcr.io/ds-artifacts-cloudshell/deploystack_custom_image&shellonly=true) +[![Open in Cloud Shell](https://gstatic.com/cloudssh/images/open-btn.png)](https://ssh.cloud.google.com/cloudshell/editor?cloudshell_git_repo=https%3A%2F%2Fgithub.com%2FBerriAI%2Flitellm&cloudshell_workspace=terraform%2Flitellm%2Fgcp%2Fexamples%2Fdefault&cloudshell_tutorial=TUTORIAL.md&cloudshell_image=gcr.io/ds-artifacts-cloudshell/deploystack_custom_image&shellonly=true) Real 1-click. Opens Cloud Shell, clones this repo, and walks you through `terraform apply` via a built-in [DeployStack tutorial](./terraform/litellm/gcp/examples/default/TUTORIAL.md) — pick the project, the tutorial sets up the Artifact Registry remote repo, writes `terraform.tfvars` from your answers, and runs apply. From 7cf1b263356fa6960da275f360db0031182fb1e2 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 7 Jun 2026 02:10:22 +0000 Subject: [PATCH 12/30] docs(readme): collapse Railway deploy anchor to a single line The Railway button wrapped its img across indented lines, so the anchor contained leading and trailing whitespace. GitHub underlines link content, rendering that whitespace as a small blue underline beside the button. Put the anchor on one line like the other three buttons so there is no inner whitespace to underline. --- README.md | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/README.md b/README.md index 9e6b4030966..719f74e5924 100644 --- a/README.md +++ b/README.md @@ -7,9 +7,7 @@

Open Source AI Gateway for 100+ LLMs. Self-hosted. Enterprise-ready. Call any LLM in OpenAI format.

Deploy to Render - - Deploy on Railway - + Deploy on Railway Deploy on AWS Deploy on GCP

From 7f57a7a068d125dbe390e665391fe63f8a26dd8e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 10 Jun 2026 03:58:08 +0000 Subject: [PATCH 13/30] Add Claude Fable 5 cost map entries as a data-only hotfix Backports only the model map changes from #30064 so deployments on released litellm versions pick up Fable 5 pricing, context window, and the adaptive thinking flag through the hosted cost map fetch without upgrading. Includes the supports_sampling_params flag on the 28 Fable 5 / Opus 4.7 / Opus 4.8 entries (ignored by released code, read by the gating that ships with the next release) and the matching one-line schema declaration so the map validation test passes. https://claude.ai/code/session_01MZarYYT3aS7DxaNjoax6Gm --- ...odel_prices_and_context_window_backup.json | 276 ++++++++++++++++++ model_prices_and_context_window.json | 276 ++++++++++++++++++ tests/test_litellm/test_utils.py | 1 + 3 files changed, 553 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 397f96fdb1e..757aacf1caf 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -1156,6 +1156,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1202,6 +1203,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1233,6 +1235,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1264,6 +1267,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1295,6 +1299,139 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh" + }, + "anthropic.claude-fable-5": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh" + }, + "global.anthropic.claude-fable-5": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh" + }, + "us.anthropic.claude-fable-5": { + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5.5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh" + }, + "eu.anthropic.claude-fable-5": { + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5.5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1327,6 +1464,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1359,6 +1497,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1391,6 +1530,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1423,6 +1563,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1455,6 +1596,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1485,6 +1627,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -2208,6 +2351,37 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_max_reasoning_effort": true + }, + "azure_ai/claude-fable-5": { + "input_cost_per_token": 1e-05, + "output_cost_per_token": 5e-05, + "litellm_provider": "azure_ai", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -2237,6 +2411,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -10133,6 +10308,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -10167,6 +10343,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -10177,6 +10354,40 @@ }, "supports_output_config": true }, + "claude-fable-5": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "anthropic", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_max_reasoning_effort": true, + "provider_specific_entry": { + "us": 1.1 + }, + "supports_output_config": true + }, "claude-opus-4-8": { "cache_creation_input_token_cost": 6.25e-06, "cache_creation_input_token_cost_above_1hr": 1e-05, @@ -10201,6 +10412,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -33967,6 +34179,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -33995,6 +34208,67 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_max_reasoning_effort": true + }, + "vertex_ai/claude-fable-5": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "vertex_ai-anthropic_models", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_max_reasoning_effort": true + }, + "vertex_ai/claude-fable-5@default": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "vertex_ai-anthropic_models", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -34024,6 +34298,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -34053,6 +34328,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index b2836a096b7..ddd7d51d76b 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -1156,6 +1156,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1202,6 +1203,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1233,6 +1235,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1264,6 +1267,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1295,6 +1299,139 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh" + }, + "anthropic.claude-fable-5": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh" + }, + "global.anthropic.claude-fable-5": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh" + }, + "us.anthropic.claude-fable-5": { + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5.5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh" + }, + "eu.anthropic.claude-fable-5": { + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5.5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1327,6 +1464,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1359,6 +1497,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1391,6 +1530,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1423,6 +1563,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1455,6 +1596,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -1485,6 +1627,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -2208,6 +2351,37 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_max_reasoning_effort": true + }, + "azure_ai/claude-fable-5": { + "input_cost_per_token": 1e-05, + "output_cost_per_token": 5e-05, + "litellm_provider": "azure_ai", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -2237,6 +2411,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -10133,6 +10308,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -10167,6 +10343,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -10177,6 +10354,40 @@ }, "supports_output_config": true }, + "claude-fable-5": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "anthropic", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_max_reasoning_effort": true, + "provider_specific_entry": { + "us": 1.1 + }, + "supports_output_config": true + }, "claude-opus-4-8": { "cache_creation_input_token_cost": 6.25e-06, "cache_creation_input_token_cost_above_1hr": 1e-05, @@ -10201,6 +10412,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -34007,6 +34219,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -34035,6 +34248,67 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_max_reasoning_effort": true + }, + "vertex_ai/claude-fable-5": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "vertex_ai-anthropic_models", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_max_reasoning_effort": true + }, + "vertex_ai/claude-fable-5@default": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "vertex_ai-anthropic_models", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_adaptive_thinking": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -34064,6 +34338,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, @@ -34093,6 +34368,7 @@ "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, + "supports_sampling_params": false, "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index f179e9c8f93..4c4d9e1133b 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -858,6 +858,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_xhigh_reasoning_effort": {"type": "boolean"}, "supports_max_reasoning_effort": {"type": "boolean"}, "supports_adaptive_thinking": {"type": "boolean"}, + "supports_sampling_params": {"type": "boolean"}, "supports_service_tier": {"type": "boolean"}, "supports_preset": {"type": "boolean"}, "supports_output_config": {"type": "boolean"}, From 15462f7d1d46981bcdae318909acbc9d587b8603 Mon Sep 17 00:00:00 2001 From: xbrxr03 Date: Sun, 21 Jun 2026 20:09:45 -0400 Subject: [PATCH 14/30] fix: correct context window tokens for GPT-5 Pro and GPT-5.4 Mini/Nano MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three bugs in model_prices_and_context_window.json: 1. gpt-5-pro and gpt-5-pro-2025-10-06: max_input_tokens and max_tokens were SWAPPED. GPT-5 Pro has a 400K context window (input) with 128K max output, but the values were set as max_input=128000, max_tokens=272000. This caused token limit errors when sending prompts over 128K tokens to GPT-5 Pro. 2. gpt-5.4-mini and gpt-5.4-mini-2026-03-17: max_input_tokens was 272000, but GPT-5.4 Mini shares the same 1,050,000 token context window as GPT-5.4. This was inconsistent with the azure/ variants which already correctly had 1,050,000. 3. gpt-5.4-nano and gpt-5.4-nano-2026-03-17: same issue as Mini, max_input_tokens was 272000 instead of 1,050,000. Source: OpenAI model documentation and contextwindows.dev which aggregates official context window sizes. Fixes #30928 (partially — the issue incorrectly claims gpt-5/gpt-5-mini should be 400K; their 272K values are correct per OpenAI docs) --- model_prices_and_context_window.json | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 3e844a8e3ed..77e968fb864 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -22388,7 +22388,7 @@ "input_cost_per_token_batches": 3.75e-07, "input_cost_per_token_priority": 1.5e-06, "litellm_provider": "openai", - "max_input_tokens": 272000, + "max_input_tokens": 1050000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", @@ -22436,7 +22436,7 @@ "input_cost_per_token_batches": 3.75e-07, "input_cost_per_token_priority": 1.5e-06, "litellm_provider": "openai", - "max_input_tokens": 272000, + "max_input_tokens": 1050000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", @@ -22482,7 +22482,7 @@ "input_cost_per_token_flex": 1e-07, "input_cost_per_token_batches": 1e-07, "litellm_provider": "openai", - "max_input_tokens": 272000, + "max_input_tokens": 1050000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", @@ -22527,7 +22527,7 @@ "input_cost_per_token_flex": 1e-07, "input_cost_per_token_batches": 1e-07, "litellm_provider": "openai", - "max_input_tokens": 272000, + "max_input_tokens": 1050000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", @@ -22568,9 +22568,9 @@ "input_cost_per_token": 1.5e-05, "input_cost_per_token_batches": 7.5e-06, "litellm_provider": "openai", - "max_input_tokens": 128000, + "max_input_tokens": 400000, "max_output_tokens": 272000, - "max_tokens": 272000, + "max_tokens": 128000, "mode": "responses", "output_cost_per_token": 0.00012, "output_cost_per_token_batches": 6e-05, @@ -22604,9 +22604,9 @@ "input_cost_per_token": 1.5e-05, "input_cost_per_token_batches": 7.5e-06, "litellm_provider": "openai", - "max_input_tokens": 128000, + "max_input_tokens": 400000, "max_output_tokens": 272000, - "max_tokens": 272000, + "max_tokens": 128000, "mode": "responses", "output_cost_per_token": 0.00012, "output_cost_per_token_batches": 6e-05, From 3d0e686de718f29f08eaa63c669d73a607f7f337 Mon Sep 17 00:00:00 2001 From: xbrxr03 Date: Mon, 22 Jun 2026 02:38:56 -0400 Subject: [PATCH 15/30] =?UTF-8?q?fix:=20also=20correct=20max=5Foutput=5Fto?= =?UTF-8?q?kens=20for=20gpt-5-pro=20(272000=E2=86=92128000)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per reviewer feedback, max_output_tokens was left at 272000 while max_tokens was corrected to 128000, causing an internal inconsistency. Both should be 128000 per OpenAI docs. --- model_prices_and_context_window.json | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 77e968fb864..22aec7e3104 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -22569,7 +22569,7 @@ "input_cost_per_token_batches": 7.5e-06, "litellm_provider": "openai", "max_input_tokens": 400000, - "max_output_tokens": 272000, + "max_output_tokens": 128000, "max_tokens": 128000, "mode": "responses", "output_cost_per_token": 0.00012, @@ -22605,7 +22605,7 @@ "input_cost_per_token_batches": 7.5e-06, "litellm_provider": "openai", "max_input_tokens": 400000, - "max_output_tokens": 272000, + "max_output_tokens": 128000, "max_tokens": 128000, "mode": "responses", "output_cost_per_token": 0.00012, From 63741e96985c1128ea53d885b605defb7d92da35 Mon Sep 17 00:00:00 2001 From: hayden Date: Wed, 24 Jun 2026 19:34:49 +0900 Subject: [PATCH 16/30] fix(cost): price gpt-image generated output tokens as image tokens (#31147) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The OpenAI Images endpoints (/v1/images/generations, /v1/images/edits) return usage with no output token breakdown — litellm's `ImageUsage` has no `output_tokens_details` field — so generated-image OUTPUT tokens were priced at the text rate (`output_cost_per_token`) instead of the image rate (`output_cost_per_image_token`). For gpt-image-2 that is $10/1M vs $30/1M, a ~3x undercount on the dominant cost component (image output is ~74% of spend). This also affects azure gpt-image, which shares this calculator. The OpenAI gpt-image cost calculator re-implemented usage handling instead of reusing `calculate_image_response_cost_from_usage`, the shared helper that azure_ai/gemini/vertex_ai already use. That helper classifies generated output tokens as image tokens when the provider does not itemize output, and splits text/image when it does. Fix: route the ImageUsage path through `calculate_image_response_cost_from_usage` (pre-transformed chat Usage objects are still costed directly). Adds a regression test for the no-breakdown ImageUsage case (gpt-image-2). --- .../image_generation/cost_calculator.py | 69 ++++++--------- .../test_gpt_image_cost_calculator.py | 88 +++++++++++++++++++ 2 files changed, 117 insertions(+), 40 deletions(-) diff --git a/litellm/llms/openai/image_generation/cost_calculator.py b/litellm/llms/openai/image_generation/cost_calculator.py index d009a085fab..dab277a7ba8 100644 --- a/litellm/llms/openai/image_generation/cost_calculator.py +++ b/litellm/llms/openai/image_generation/cost_calculator.py @@ -7,7 +7,10 @@ These models use token-based pricing instead of pixel-based pricing like DALL-E. from typing import Optional from litellm import verbose_logger -from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token +from litellm.litellm_core_utils.llm_cost_calc.utils import ( + calculate_image_response_cost_from_usage, + generic_cost_per_token, +) from litellm.types.utils import ImageResponse, Usage @@ -16,54 +19,40 @@ def cost_calculator( image_response: ImageResponse, custom_llm_provider: Optional[str] = None, ) -> float: - """ - Calculate cost for OpenAI gpt-image models. - - Uses the same usage format as Responses API, so we reuse the helper - to transform to chat completion format and use generic_cost_per_token. - - Args: - model: The model name (e.g., "gpt-image-1", "gpt-image-2") - image_response: The ImageResponse containing usage data - custom_llm_provider: Optional provider name - - Returns: - float: Total cost in USD - """ + """Calculate cost for OpenAI gpt-image models (token-based pricing).""" usage = getattr(image_response, "usage", None) - if usage is None: verbose_logger.debug( f"No usage data available for {model}, cannot calculate token-based cost" ) return 0.0 - # If usage is already a Usage object with completion_tokens_details set, - # use it directly (it was already transformed in convert_to_image_response) + provider = custom_llm_provider or "openai" + + # A chat Usage with an explicit output breakdown: cost via generic_cost_per_token. if isinstance(usage, Usage) and usage.completion_tokens_details is not None: - chat_usage = usage - else: - # Transform ImageUsage to Usage using the existing helper - # ImageUsage has the same format as ResponseAPIUsage - from litellm.responses.utils import ResponseAPILoggingUtils - - chat_usage = ( - ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(usage) + prompt_cost, completion_cost = generic_cost_per_token( + model=model, usage=usage, custom_llm_provider=provider ) + return prompt_cost + completion_cost - # Use generic_cost_per_token for cost calculation - prompt_cost, completion_cost = generic_cost_per_token( - model=model, - usage=chat_usage, - custom_llm_provider=custom_llm_provider or "openai", - ) + # ImageUsage / ResponseAPIUsage: reuse the shared helper (same path as + # azure_ai/gemini/vertex_ai). It prices generated output tokens at + # output_cost_per_image_token, classifying them as image tokens when the provider + # does not itemize output and splitting text/image when it does. + if getattr(usage, "input_tokens", None) is not None: + token_based_cost = calculate_image_response_cost_from_usage( + model=model, image_response=image_response, custom_llm_provider=provider + ) + if token_based_cost is not None: + return token_based_cost - total_cost = prompt_cost + completion_cost + # Fallback: a Usage with no output breakdown that the image helper can't read — + # cost via generic_cost_per_token (text rate) instead of returning 0.0. + if isinstance(usage, Usage): + prompt_cost, completion_cost = generic_cost_per_token( + model=model, usage=usage, custom_llm_provider=provider + ) + return prompt_cost + completion_cost - verbose_logger.debug( - f"OpenAI gpt-image cost calculation for {model}: " - f"prompt_cost=${prompt_cost:.6f}, completion_cost=${completion_cost:.6f}, " - f"total=${total_cost:.6f}" - ) - - return total_cost + return 0.0 diff --git a/tests/test_litellm/test_gpt_image_cost_calculator.py b/tests/test_litellm/test_gpt_image_cost_calculator.py index 6644b1389cf..c371f7442be 100644 --- a/tests/test_litellm/test_gpt_image_cost_calculator.py +++ b/tests/test_litellm/test_gpt_image_cost_calculator.py @@ -381,5 +381,93 @@ class TestCompletionCostIntegration: assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}" +class TestGPTImage2OutputImageTokensNoBreakdown: + """ + Regression test: the OpenAI Images endpoints (/v1/images/generations and + /v1/images/edits) return usage with NO output token breakdown — litellm's + ImageUsage has no ``output_tokens_details`` field. Before the fix, the + generated-image OUTPUT tokens were priced at the text rate + (``output_cost_per_token`` = $10/1M for gpt-image-2) instead of the image rate + (``output_cost_per_image_token`` = $30/1M), a ~3x undercount on the dominant + cost component. + """ + + def test_gpt_image_2_output_priced_as_image_when_no_breakdown(self): + from litellm.llms.openai.image_generation.cost_calculator import ( + cost_calculator, + ) + + # Mirrors a real gpt-image-2 /v1/images/edits response: input breakdown is + # present, but there is no usable output token breakdown. + usage = ImageUsage( + input_tokens=3987, + output_tokens=5488, + total_tokens=9475, + input_tokens_details=ImageUsageInputTokensDetails( + text_tokens=943, + image_tokens=3044, + ), + ) + + image_response = ImageResponse( + created=1234567890, + data=[ImageObject(b64_json="test")], + ) + image_response.usage = usage + image_response._hidden_params = {"custom_llm_provider": "openai"} + + cost = cost_calculator( + model="gpt-image-2", + image_response=image_response, + custom_llm_provider="openai", + ) + + # gpt-image-2 pricing: + # text input: 943 * $5/1M = 0.004715 + # image input: 3044 * $8/1M = 0.024352 + # image output: 5488 * $30/1M = 0.164640 (NOT text output $10/1M = 0.054880) + expected_cost = 943 * 5e-6 + 3044 * 8e-6 + 5488 * 3e-5 + assert abs(cost - expected_cost) < 1e-6, ( + f"Expected {expected_cost}, got {cost}. Generated image output tokens " + f"are likely being priced at the text output_cost_per_token rate." + ) + + def test_gpt_image_2_chat_usage_without_breakdown_is_costed_not_zero(self): + """A chat ``Usage`` with ``completion_tokens_details=None`` must still be + costed via ``generic_cost_per_token`` (output at the text rate) rather than + erroring or silently returning 0.0.""" + from litellm.llms.openai.image_generation.cost_calculator import ( + cost_calculator, + ) + + usage = Usage( + prompt_tokens=600, + completion_tokens=5000, + total_tokens=5600, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=100, + image_tokens=500, + ), + ) + + image_response = ImageResponse( + created=1234567890, + data=[ImageObject(b64_json="test")], + ) + image_response.usage = usage + image_response._hidden_params = {"custom_llm_provider": "openai"} + + cost = cost_calculator( + model="gpt-image-2", + image_response=image_response, + custom_llm_provider="openai", + ) + + # No output breakdown -> output priced at the text rate (output_cost_per_token): + # text in 100*$5/1M + image in 500*$8/1M + output 5000*$10/1M + expected_cost = 100 * 5e-6 + 500 * 8e-6 + 5000 * 1e-5 + assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}" + + if __name__ == "__main__": pytest.main([__file__, "-v"]) From 6893df16fbd11649d0910df95211ea399ce6ca1e Mon Sep 17 00:00:00 2001 From: Kent <72616338+kingdoooo@users.noreply.github.com> Date: Wed, 24 Jun 2026 18:38:20 +0800 Subject: [PATCH 17/30] fix(bedrock): route application-inference-profile ARNs to converse (#18258) (#31098) A bare application-inference-profile ARN passed as bedrock/arn:... fell through to the invoke route, which cannot derive a provider from the opaque profile id and raised 'Unknown provider=None'. The converse route needs no provider, so detect these ARNs in get_bedrock_route and route them to converse, matching the behavior of the already-documented bedrock/converse/arn:... workaround. Explicit invoke/ prefixes still win, and they remain a dead end for these ARNs by design (no provider derivable). System-defined inference-profile ARNs that embed a known model, and other opaque ARN types (provisioned-model, imported-model, custom-model-deployment) that are frequently invoke-only, are deliberately left on their current routes; tests guard both boundaries. --- litellm/llms/bedrock/common_utils.py | 12 ++++ .../llms/bedrock/test_bedrock_common_utils.py | 62 +++++++++++++++++++ 2 files changed, 74 insertions(+) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 5e97394f459..0c4e7cf2568 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -601,6 +601,15 @@ def extract_model_name_from_bedrock_arn(model: str) -> str: return model +def is_bedrock_application_inference_profile_arn(model: str) -> bool: + """ + An application inference profile ARN ends in an opaque id with no provider + substring, so the invoke path cannot resolve a provider from it. Such ARNs + must use the converse route, which needs no provider. + """ + return ":application-inference-profile/" in model + + def strip_bedrock_routing_prefix(model: str) -> str: """Strip LiteLLM routing prefixes from model name.""" for prefix in ["bedrock/", "converse/", "invoke/", "openai/", "nova-2/", "nova/"]: @@ -916,6 +925,9 @@ class BedrockModelInfo(BaseLLMModelInfo): ) or _model_after_bedrock.startswith("nova/"): return "converse" + if is_bedrock_application_inference_profile_arn(model): + return "converse" + base_model = BedrockModelInfo.get_base_model(model) alt_model = BedrockModelInfo.get_non_litellm_routing_model_name(model=model) if ( diff --git a/tests/test_litellm/llms/bedrock/test_bedrock_common_utils.py b/tests/test_litellm/llms/bedrock/test_bedrock_common_utils.py index 6298eeb25e9..3893ca474db 100644 --- a/tests/test_litellm/llms/bedrock/test_bedrock_common_utils.py +++ b/tests/test_litellm/llms/bedrock/test_bedrock_common_utils.py @@ -158,6 +158,68 @@ def test_deepseek_cris(): assert bedrock_route == "converse" +def test_application_inference_profile_arn_routes_to_converse(): + """ + Regression for #18258: a bare application-inference-profile ARN passed as + `bedrock/arn:...` must route to converse. The ARN ends in an opaque id with + no provider substring, so the invoke path cannot build a provider-native + body and raises "Unknown provider=None". Converse needs no provider, so it + is the correct route. + """ + route = BedrockModelInfo.get_bedrock_route( + model="bedrock/arn:aws:bedrock:us-west-2:123412341234:application-inference-profile/a1b2c3" + ) + assert route == "converse" + + +def test_explicit_invoke_prefix_wins_over_application_inference_profile_arn(): + """ + An explicit invoke/ prefix is respected even for an application-inference-profile + ARN; only the bare `bedrock/arn:...` form is auto-routed to converse. The + explicit invoke path remains a dead end for these ARNs (no provider can be + derived, so completion raises "Unknown provider=None") by design: a caller + that explicitly asks for invoke gets invoke. The auto-route only rescues the + documented bare form. + """ + from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM + + model = "bedrock/invoke/arn:aws:bedrock:us-west-2:123412341234:application-inference-profile/a1b2c3" + assert BedrockModelInfo.get_bedrock_route(model) == "invoke" + assert BaseAWSLLM.get_bedrock_invoke_provider(model) is None + + +def test_system_defined_inference_profile_arn_still_routes_to_converse(): + """ + A system-defined cross-region inference-profile ARN embeds a known model, so + get_base_model resolves it and it already routes to converse. Guards that the + application-inference-profile fix does not change this working case. + """ + route = BedrockModelInfo.get_bedrock_route( + model="bedrock/arn:aws:bedrock:us-east-1:123:inference-profile/us.anthropic.claude-3-5-sonnet-20240620-v1:0" + ) + assert route == "converse" + + +def test_other_opaque_arn_types_still_route_to_invoke(): + """ + Only application-inference-profile ARNs are auto-routed to converse. Other + opaque ARNs (provisioned-model, imported-model, custom-model-deployment) + also yield no invoke provider, but they are frequently invoke-only with + provider-specific body formats, so routing them to converse could break + them. Guards the deliberate scope against an over-broad "any opaque ARN -> + converse" generalization. + """ + for arn_segment in ( + "provisioned-model/abcdefgh1234", + "imported-model/abcdefgh1234", + "custom-model-deployment/abcdefgh1234", + ): + route = BedrockModelInfo.get_bedrock_route( + model=f"bedrock/arn:aws:bedrock:us-east-1:123412341234:{arn_segment}" + ) + assert route == "invoke", f"{arn_segment} should stay on invoke route" + + def test_govcloud_cross_region_inference_prefix(): """ Test that GovCloud models with cross-region inference prefix (us-gov.) are parsed correctly From e71d6ef8baf2f628be3b3a2f5e6d6d49d18d6a4f Mon Sep 17 00:00:00 2001 From: Wassim Badraoui <98709649+Wassbdr@users.noreply.github.com> Date: Wed, 24 Jun 2026 12:40:52 +0200 Subject: [PATCH 18/30] fix(moonshot): stop mutating caller messages on tool_choice='required' (#31060) _add_tool_choice_required_message appended the "select a tool" prompt to the caller's messages list in place, so transform_request corrupted the caller's conversation history and appended a duplicate prompt on every retry. Build and return a new list instead so the call stays idempotent. Adds a regression test asserting the input messages list is unchanged across repeated transform_request calls. Co-authored-by: Wassbdr --- litellm/llms/moonshot/chat/transformation.py | 10 +++---- .../test_moonshot_chat_transformation.py | 28 +++++++++++++++++++ 2 files changed, 33 insertions(+), 5 deletions(-) diff --git a/litellm/llms/moonshot/chat/transformation.py b/litellm/llms/moonshot/chat/transformation.py index da8687bce72..9399ca88583 100644 --- a/litellm/llms/moonshot/chat/transformation.py +++ b/litellm/llms/moonshot/chat/transformation.py @@ -238,11 +238,11 @@ class MoonshotChatConfig(OpenAIGPTConfig): https://platform.moonshot.ai/docs/guide/migrating-from-openai-to-kimi#about-tool_choice """ - messages.append( + optional_params.pop("tool_choice") + return [ + *messages, { "role": "user", "content": "Please select a tool to handle the current issue.", # Usually, the Kimi large language model understands the intention to invoke a tool and selects one for invocation - } - ) - optional_params.pop("tool_choice") - return messages + }, + ] diff --git a/tests/test_litellm/llms/moonshot/test_moonshot_chat_transformation.py b/tests/test_litellm/llms/moonshot/test_moonshot_chat_transformation.py index 95ade4290e9..417dd4a767c 100644 --- a/tests/test_litellm/llms/moonshot/test_moonshot_chat_transformation.py +++ b/tests/test_litellm/llms/moonshot/test_moonshot_chat_transformation.py @@ -305,6 +305,34 @@ class TestMoonshotConfig: assert len(result["messages"]) == 2 assert result["messages"][1]["content"] == "Please select a tool to handle the current issue." + def test_tool_choice_required_does_not_mutate_input_messages(self): + """tool_choice='required' must not mutate the caller's messages list. + + The handling appends a "select a tool" user message; building it in + place corrupts the caller's conversation history and makes + transform_request non-idempotent across retries. + """ + config = MoonshotChatConfig() + + messages = [{"role": "user", "content": "What's the weather like?"}] + + for _ in range(2): + optional_params = { + "tool_choice": "required", + "tools": [{"type": "function", "function": {"name": "get_weather"}}], + } + result = config.transform_request( + model="moonshot-v1-8k", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + # The returned request carries the extra message. + assert len(result["messages"]) == 2 + # The caller's list is untouched, so repeated calls stay idempotent. + assert messages == [{"role": "user", "content": "What's the weather like?"}] + def test_tool_choice_non_required_preserved(self): """Test that non-'required' tool_choice values are preserved""" config = MoonshotChatConfig() From 00725de9f2c16f8d8fb4d91e1ae171f7c8a60867 Mon Sep 17 00:00:00 2001 From: Neimar Avila Date: Wed, 24 Jun 2026 07:43:12 -0300 Subject: [PATCH 19/30] fix(transcription): accept fractional usage.seconds in diarized_json responses (#30996) gpt-4o-transcribe and compatible ASR backends return a diarized_json response with usage={"type": "duration", "seconds": }, e.g. 295.8. TranscriptionUsageDurationObject typed seconds as int, so parsing the response raised a pydantic ValidationError (int_from_float). That error surfaces as an APIConnectionError which the router treats as retryable, so it keeps re-calling the upstream (200 every time) until the upstream rate-limits and returns 429 to the caller. OpenAI specs this field as a float (see openai SDK UsageDuration.seconds), so widen seconds to float. With the parse succeeding there is no exception left to retry, which removes the loop. Co-authored-by: Neimar Avila <19142978+neimaravila@users.noreply.github.com> --- litellm/types/utils.py | 2 +- .../test_transcription_duration_hidden.py | 47 ++++++++++++++++++- 2 files changed, 47 insertions(+), 2 deletions(-) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 24d6e84fba7..d28a2680b73 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2441,7 +2441,7 @@ class ImageResponse(OpenAIImageResponse, BaseLiteLLMOpenAIResponseObject): class TranscriptionUsageDurationObject(BaseModel): type: Literal["duration"] - seconds: int + seconds: float class TranscriptionUsageInputTokenDetailsObject(BaseModel): diff --git a/tests/test_litellm/llms/openai/transcriptions/test_transcription_duration_hidden.py b/tests/test_litellm/llms/openai/transcriptions/test_transcription_duration_hidden.py index 2b287e456a1..703fa13cbc9 100644 --- a/tests/test_litellm/llms/openai/transcriptions/test_transcription_duration_hidden.py +++ b/tests/test_litellm/llms/openai/transcriptions/test_transcription_duration_hidden.py @@ -13,7 +13,52 @@ from litellm.cost_calculator import completion_cost from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( convert_to_model_response_object, ) -from litellm.types.utils import TranscriptionResponse +from litellm.types.utils import ( + TranscriptionResponse, + TranscriptionUsageDurationObject, +) + + +class TestDiarizedJsonUsageParsing: + """gpt-4o-transcribe / diarized_json returns a fractional `usage.seconds`.""" + + def test_fractional_duration_seconds_does_not_raise(self): + """ + A diarized_json response carries usage={"type": "duration", "seconds": }. + OpenAI specs `seconds` as a float, so a fractional value must parse cleanly + instead of raising and getting retried until the upstream rate-limits. + """ + response_object = { + "text": "speaker_1: Olá", + "task": "transcribe", + "duration": 295.8, + "segments": [ + { + "id": "seg_001", + "speaker": "speaker_1", + "start": 0.0, + "end": 1.0, + "text": "Olá", + "type": "transcript.text.segment", + } + ], + "usage": {"type": "duration", "seconds": 295.8}, + } + + result = convert_to_model_response_object( + response_object=response_object, + model_response_object=TranscriptionResponse(), + response_type="audio_transcription", + ) + + assert isinstance(result.usage, TranscriptionUsageDurationObject) + assert result.usage.seconds == 295.8 + + def test_usage_duration_object_accepts_float_seconds(self): + assert ( + TranscriptionUsageDurationObject(type="duration", seconds=295.8).seconds + == 295.8 + ) class TestTranscriptionDurationNotInResponseBody: From 04c649247610d1de14f4a10b5537e072014784a1 Mon Sep 17 00:00:00 2001 From: Jerry-Scintilla Date: Wed, 24 Jun 2026 18:49:19 +0800 Subject: [PATCH 20/30] fix(deepseek): drop non-function tools before chat completions call (#30910) * fix(deepseek): drop non-function tools before chat completions call DeepSeek's /chat/completions only accepts tools of type "function". Requests bridged from /v1/responses can carry responses-API-native tool types, for example a Codex CLI tool typed "namespace", which DeepSeek rejects with "unknown variant 'namespace', expected 'function'" so the whole request fails (issue #30722). Filter unsupported tool types in the DeepSeek request transform so the function tools still go through; when nothing callable remains, also drop the now-dangling tool_choice and parallel_tool_calls Fixes #30722 * test(deepseek): cover async tool filtering and document tool_choice assumption Add an async_transform_request regression test so the sync and async tool filtering paths cannot silently diverge, and document in _drop_unsupported_tools that only non-function tools are dropped, so a function-named tool_choice always references a surviving tool --- litellm/llms/deepseek/chat/transformation.py | 52 +++++++++ .../llms/deepseek/chat/__init__.py | 0 .../chat/test_deepseek_chat_transformation.py | 103 ++++++++++++++++++ 3 files changed, 155 insertions(+) create mode 100644 tests/test_litellm/llms/deepseek/chat/__init__.py create mode 100644 tests/test_litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index 7ed3e484535..a316a3b9260 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -146,6 +146,56 @@ class DeepSeekChatConfig(OpenAIGPTConfig): and (optional_params.get("thinking") or {}).get("type") == "enabled" ) + @staticmethod + def _drop_unsupported_tools(optional_params: dict) -> dict: + """ + DeepSeek's /chat/completions only accepts tools of type "function". + + Requests bridged from /v1/responses can carry responses-API-native tool + types (e.g. a Codex CLI tool typed "namespace"); DeepSeek rejects the + whole request with `unknown variant '', expected 'function'` (issue + #30722). Drop the unsupported entries so the function tools still go + through, and drop the now-dangling tool_choice/parallel_tool_calls when + nothing callable survives. + + Only non-`function` tools are ever dropped, so a `tool_choice` that names + a specific function still points at a surviving tool and is left intact; + `tool_choice`/`parallel_tool_calls` are cleared only when no function + tool remains. + """ + tools = optional_params.get("tools") + if not isinstance(tools, list) or not tools: + return optional_params + + def _is_function_tool(tool: object) -> bool: + return isinstance(tool, dict) and tool.get("type") == "function" + + function_tools = [tool for tool in tools if _is_function_tool(tool)] + if len(function_tools) == len(tools): + return optional_params + + dropped_types = sorted( + { + str(tool.get("type")) if isinstance(tool, dict) else type(tool).__name__ + for tool in tools + if not _is_function_tool(tool) + } + ) + litellm.verbose_logger.warning( + "DeepSeek chat completions only supports function tools; dropping " + "unsupported tool type(s) %s before sending the request", + dropped_types, + ) + + cleaned = {k: v for k, v in optional_params.items() if k != "tools"} + if function_tools: + return {**cleaned, "tools": function_tools} + return { + k: v + for k, v in cleaned.items() + if k not in ("tool_choice", "parallel_tool_calls") + } + def transform_request( self, model: str, @@ -163,6 +213,7 @@ class DeepSeekChatConfig(OpenAIGPTConfig): (user explicitly enabled it), preventing spurious injection on models like deepseek-v3.2 that support thinking as opt-in but not always-on. """ + optional_params = self._drop_unsupported_tools(optional_params) if self._thinking_mode_active(model=model, optional_params=optional_params): messages = self._fill_reasoning_content(messages) return super().transform_request( @@ -185,6 +236,7 @@ class DeepSeekChatConfig(OpenAIGPTConfig): Async equivalent of transform_request — applies the same reasoning_content fix for multi-turn thinking-mode conversations. """ + optional_params = self._drop_unsupported_tools(optional_params) if self._thinking_mode_active(model=model, optional_params=optional_params): messages = self._fill_reasoning_content(messages) return await super().async_transform_request( diff --git a/tests/test_litellm/llms/deepseek/chat/__init__.py b/tests/test_litellm/llms/deepseek/chat/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py b/tests/test_litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py new file mode 100644 index 00000000000..ec51e5d303d --- /dev/null +++ b/tests/test_litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py @@ -0,0 +1,103 @@ +from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig + + +def _function_tool(name: str) -> dict: + return { + "type": "function", + "function": {"name": name, "parameters": {"type": "object"}}, + } + + +def test_drop_unsupported_tools_keeps_function_tools_only(): + optional_params = { + "tools": [ + _function_tool("shell"), + {"type": "namespace", "name": "container.exec"}, + _function_tool("apply_patch"), + ], + "tool_choice": "auto", + } + + result = DeepSeekChatConfig._drop_unsupported_tools(optional_params) + + assert [tool["function"]["name"] for tool in result["tools"]] == [ + "shell", + "apply_patch", + ] + assert all(tool["type"] == "function" for tool in result["tools"]) + assert result["tool_choice"] == "auto" + + +def test_drop_unsupported_tools_drops_dangling_tool_choice_when_none_survive(): + optional_params = { + "tools": [{"type": "namespace", "name": "container.exec"}], + "tool_choice": "required", + "parallel_tool_calls": True, + "temperature": 0.2, + } + + result = DeepSeekChatConfig._drop_unsupported_tools(optional_params) + + assert "tools" not in result + assert "tool_choice" not in result + assert "parallel_tool_calls" not in result + assert result["temperature"] == 0.2 + + +def test_drop_unsupported_tools_is_noop_for_function_only(): + optional_params = { + "tools": [_function_tool("shell")], + "tool_choice": "auto", + } + + result = DeepSeekChatConfig._drop_unsupported_tools(optional_params) + + assert result is optional_params + + +def test_drop_unsupported_tools_is_noop_without_tools(): + optional_params = {"temperature": 0.7} + + result = DeepSeekChatConfig._drop_unsupported_tools(optional_params) + + assert result is optional_params + + +def test_transform_request_strips_unsupported_tools_from_body(): + config = DeepSeekChatConfig() + body = config.transform_request( + model="deepseek-chat", + messages=[{"role": "user", "content": "hi"}], + optional_params={ + "tools": [ + _function_tool("shell"), + {"type": "namespace", "name": "container.exec"}, + ], + "tool_choice": "auto", + }, + litellm_params={}, + headers={}, + ) + + assert [tool["type"] for tool in body["tools"]] == ["function"] + assert body["tools"][0]["function"]["name"] == "shell" + + +async def test_async_transform_request_strips_unsupported_tools_from_body(): + config = DeepSeekChatConfig() + body = await config.async_transform_request( + model="deepseek-chat", + messages=[{"role": "user", "content": "hi"}], + optional_params={ + "tools": [ + _function_tool("shell"), + {"type": "namespace", "name": "container.exec"}, + ], + "tool_choice": "auto", + }, + litellm_params={}, + headers={}, + ) + + assert [tool["type"] for tool in body["tools"]] == ["function"] + assert body["tools"][0]["function"]["name"] == "shell" From d7205918b51b07026acfbbecd16647f63b2729ba Mon Sep 17 00:00:00 2001 From: AlexBGoode Date: Wed, 24 Jun 2026 13:52:09 +0300 Subject: [PATCH 21/30] feat(catalog): add zai/glm-5.1, zai/glm-4.7-flash, openrouter/z-ai/glm-5.1 (#29840) --- model_prices_and_context_window.json | 46 ++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 5f3f2294147..d7017a40993 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -29866,6 +29866,22 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "openrouter/z-ai/glm-5.1": { + "input_cost_per_token": 1.05e-06, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 5.25e-07, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "openrouter", + "max_input_tokens": 202752, + "max_output_tokens": 65535, + "max_tokens": 65535, + "mode": "chat", + "source": "https://openrouter.ai/z-ai/glm-5.1", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, "openrouter/minimax/minimax-m2.1": { "input_cost_per_token": 2.7e-07, "output_cost_per_token": 1.2e-06, @@ -37872,6 +37888,21 @@ "supports_tool_choice": true, "source": "https://docs.z.ai/guides/overview/pricing" }, + "zai/glm-5.1": { + "cache_creation_input_token_cost": 0, + "cache_read_input_token_cost": 2.6e-07, + "input_cost_per_token": 1.4e-06, + "output_cost_per_token": 4.4e-06, + "litellm_provider": "zai", + "max_input_tokens": 200000, + "max_output_tokens": 128000, + "mode": "chat", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "source": "https://docs.z.ai/guides/overview/pricing" + }, "zai/glm-5-code": { "cache_creation_input_token_cost": 0, "cache_read_input_token_cost": 3e-07, @@ -37902,6 +37933,21 @@ "supports_tool_choice": true, "source": "https://docs.z.ai/guides/overview/pricing" }, + "zai/glm-4.7-flash": { + "cache_creation_input_token_cost": 0, + "cache_read_input_token_cost": 0, + "input_cost_per_token": 0, + "output_cost_per_token": 0, + "litellm_provider": "zai", + "max_input_tokens": 200000, + "max_output_tokens": 128000, + "mode": "chat", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "source": "https://docs.z.ai/guides/overview/pricing" + }, "zai/glm-4.6": { "cache_creation_input_token_cost": 0, "cache_read_input_token_cost": 1.1e-07, From 6bffdebd1d75e2191a055f3415eef24cd89f01b9 Mon Sep 17 00:00:00 2001 From: Carsten Boloz Date: Wed, 24 Jun 2026 06:53:34 -0400 Subject: [PATCH 22/30] feat(ui): surface team budget on key overview when key has no own budget (#30801) * feat(ui): surface team budget on key overview when key has no own budget * fix(ui): replace IIFE with derived variable and use find() for team budget display --- .../VirtualKeysPage/VirtualKeysTable.tsx | 11 ++- .../key_info_view.budget_display.test.tsx | 79 ++++++++++++++++++- .../components/templates/key_info_view.tsx | 16 ++-- 3 files changed, 95 insertions(+), 11 deletions(-) diff --git a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx index 299f8a05f71..0c95d5dcfb7 100644 --- a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx +++ b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx @@ -467,10 +467,15 @@ export function VirtualKeysTable({ teams, organizations, onSortChange, currentSo enableSorting: true, cell: (info) => { const maxBudget = info.getValue() as number | null; - if (maxBudget === null) { - return "Unlimited"; + if (maxBudget !== null) { + return `$${formatNumberWithCommas(maxBudget)}`; } - return `$${formatNumberWithCommas(maxBudget)}`; + const teamId = info.row.original.team_id; + const team = teams?.find((t) => t.team_id === teamId); + if (team?.max_budget != null) { + return `$${formatNumberWithCommas(team.max_budget)} (Team)`; + } + return "Unlimited"; }, }, { diff --git a/ui/litellm-dashboard/src/components/templates/key_info_view.budget_display.test.tsx b/ui/litellm-dashboard/src/components/templates/key_info_view.budget_display.test.tsx index 77abde3d870..5407d37fcf1 100644 --- a/ui/litellm-dashboard/src/components/templates/key_info_view.budget_display.test.tsx +++ b/ui/litellm-dashboard/src/components/templates/key_info_view.budget_display.test.tsx @@ -1,7 +1,7 @@ import { renderWithProviders } from "../../../tests/test-utils"; import { screen, waitFor } from "@testing-library/react"; import { beforeEach, describe, expect, it, vi } from "vitest"; -import { KeyResponse } from "../key_team_helpers/key_list"; +import { KeyResponse, Team } from "../key_team_helpers/key_list"; import KeyInfoView from "./key_info_view"; import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized"; import useTeams from "@/app/(dashboard)/hooks/useTeams"; @@ -103,8 +103,26 @@ const baseAuthorized = { userEmail: null, disabledPersonalKeyCreation: null, showSSOBanner: false, + isLoading: false, + isAuthorized: true, }; +const makeTeam = (overrides: Partial): Team => ({ + team_id: "team-default", + team_alias: "Default Team", + models: [], + max_budget: null, + budget_duration: null, + tpm_limit: null, + rpm_limit: null, + organization_id: "", + created_at: "2026-01-01T00:00:00Z", + keys: [], + members_with_roles: [], + spend: 0, + ...overrides, +}); + describe("KeyInfoView overview budget display (LIT-2845)", () => { beforeEach(() => { vi.mocked(useTeams).mockReturnValue({ teams: [], setTeams: vi.fn() }); @@ -151,7 +169,64 @@ describe("KeyInfoView overview budget display (LIT-2845)", () => { it("renders 'Unlimited' when max_budget is null", async () => { renderWithProviders( {}} + keyId={"test-key-id"} + onKeyDataUpdate={() => {}} + teams={[]} + />, + ); + await waitFor(() => { + expect(screen.getByText(/of Unlimited/)).toBeInTheDocument(); + }); + }); + + it("renders team budget with alias and duration when key has no own budget but team has one", async () => { + vi.mocked(useTeams).mockReturnValue({ + teams: [makeTeam({ team_id: "team-123", team_alias: "Test Budget", max_budget: 1200, budget_duration: "30d" })], + setTeams: vi.fn(), + }); + renderWithProviders( + {}} + keyId={"test-key-id"} + onKeyDataUpdate={() => {}} + teams={[]} + />, + ); + await waitFor(() => { + expect(screen.getByText(/of \$1,200\.00 \(Team: Test Budget \/ 30d\)/)).toBeInTheDocument(); + }); + }); + + it("renders team budget without duration when team has no budget_duration", async () => { + vi.mocked(useTeams).mockReturnValue({ + teams: [makeTeam({ team_id: "team-456", team_alias: "No Duration Team", max_budget: 500 })], + setTeams: vi.fn(), + }); + renderWithProviders( + {}} + keyId={"test-key-id"} + onKeyDataUpdate={() => {}} + teams={[]} + />, + ); + await waitFor(() => { + expect(screen.getByText(/of \$500\.00 \(Team: No Duration Team\)/)).toBeInTheDocument(); + }); + }); + + it("renders 'Unlimited' when key has no budget and team also has no budget", async () => { + vi.mocked(useTeams).mockReturnValue({ + teams: [makeTeam({ team_id: "team-789", team_alias: "Free Team" })], + setTeams: vi.fn(), + }); + renderWithProviders( + {}} keyId={"test-key-id"} onKeyDataUpdate={() => {}} diff --git a/ui/litellm-dashboard/src/components/templates/key_info_view.tsx b/ui/litellm-dashboard/src/components/templates/key_info_view.tsx index 018880b70aa..4244ae2d794 100644 --- a/ui/litellm-dashboard/src/components/templates/key_info_view.tsx +++ b/ui/litellm-dashboard/src/components/templates/key_info_view.tsx @@ -411,6 +411,15 @@ export default function KeyInfoView({ }); }; + const parentTeam = currentKeyData.team_id ? teamsData?.find((team) => team.team_id === currentKeyData.team_id) : null; + + const budgetDisplay = + currentKeyData.max_budget !== null + ? `$${formatNumberWithCommas(currentKeyData.max_budget, 2)}` + : parentTeam?.max_budget != null + ? `$${formatNumberWithCommas(parentTeam.max_budget, 2)} (Team: ${parentTeam.team_alias || parentTeam.team_id}${parentTeam.budget_duration ? ` / ${parentTeam.budget_duration}` : ""})` + : "Unlimited"; + return (
Spend
${formatNumberWithCommas(currentKeyData.spend, 4)} - - of{" "} - {currentKeyData.max_budget !== null - ? `$${formatNumberWithCommas(currentKeyData.max_budget, 2)}` - : "Unlimited"} - + of {budgetDisplay}
From 4b22aa1fcaf9e7b07ba4396146b2d5aa34e453d9 Mon Sep 17 00:00:00 2001 From: jesco Date: Wed, 24 Jun 2026 07:03:52 -0400 Subject: [PATCH 23/30] fix(anthropic): emit replayable streaming thinking blocks (#31022) --- litellm/llms/anthropic/chat/handler.py | 59 ++++---- .../chat/test_anthropic_chat_handler.py | 131 ++++++++++++++++++ 2 files changed, 159 insertions(+), 31 deletions(-) diff --git a/litellm/llms/anthropic/chat/handler.py b/litellm/llms/anthropic/chat/handler.py index 5d14f3cc4ae..d2e5b89371e 100644 --- a/litellm/llms/anthropic/chat/handler.py +++ b/litellm/llms/anthropic/chat/handler.py @@ -629,6 +629,7 @@ class ModelResponseIterator: Optional[ChatCompletionToolCallChunk], List[Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]], Dict[str, Any], + Optional[str], ]: """ Helper function to handle the content block delta @@ -636,6 +637,7 @@ class ModelResponseIterator: text = "" tool_use: Optional[ChatCompletionToolCallChunk] = None provider_specific_fields = {} + reasoning_content: Optional[str] = None content_block = ContentBlockDelta(**chunk) # type: ignore thinking_blocks: List[ Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock] @@ -670,14 +672,24 @@ class ModelResponseIterator: thinking_content = content_block["delta"].get("thinking") if isinstance(thinking_content, str) and thinking_content: self.reasoning_content_chunks.append(thinking_content) - thinking_blocks = [ - ChatCompletionThinkingBlock( - type="thinking", - thinking=thinking_content or "", - signature=str(content_block["delta"].get("signature") or ""), - ) - ] - provider_specific_fields["thinking_blocks"] = thinking_blocks + reasoning_content = thinking_content + + signature = content_block["delta"].get("signature") + if isinstance(signature, str) and signature: + thinking_blocks = [ + ChatCompletionThinkingBlock( + type="thinking", + thinking="".join( + cast(str, block["delta"].get("thinking")) + for block in self.content_blocks + if isinstance(block["delta"].get("thinking"), str) + ), + signature=signature, + ) + ] + provider_specific_fields["thinking_blocks"] = thinking_blocks + if reasoning_content is None: + reasoning_content = "" elif ( "content" in content_block["delta"] and content_block["delta"].get("type") == "compaction_delta" @@ -688,25 +700,13 @@ class ModelResponseIterator: "content": content_block["delta"]["content"], } - return text, tool_use, thinking_blocks, provider_specific_fields - - def _handle_reasoning_content( - self, - thinking_blocks: List[ - Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock] - ], - ) -> Optional[str]: - """ - Handle the reasoning content - """ - reasoning_content = None - for block in thinking_blocks: - thinking_content = cast(Optional[str], block.get("thinking")) - if reasoning_content is None: - reasoning_content = "" - if thinking_content is not None: - reasoning_content += thinking_content - return reasoning_content + return ( + text, + tool_use, + thinking_blocks, + provider_specific_fields, + reasoning_content, + ) def _handle_redacted_thinking_content( self, @@ -802,11 +802,8 @@ class ModelResponseIterator: tool_use, thinking_blocks, provider_specific_fields, + reasoning_content, ) = self._content_block_delta_helper(chunk=chunk) - if thinking_blocks: - reasoning_content = self._handle_reasoning_content( - thinking_blocks=thinking_blocks - ) elif type_chunk == "content_block_start": """ event: content_block_start diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_handler.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_handler.py index 2fdd639e74d..0b1aaf87516 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_handler.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_handler.py @@ -74,6 +74,137 @@ def test_redacted_thinking_content_block_delta(): assert "thinking_blocks" in model_response.choices[0].delta.provider_specific_fields +def test_streaming_thinking_blocks_are_replayable_after_signature_delta(): + model_response_iterator = ModelResponseIterator( + streaming_response=MagicMock(), sync_stream=True, json_mode=False + ) + chunks = [ + { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "thinking", "thinking": ""}, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": "Step 1. "}, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": "Step 2."}, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "signature_delta", "signature": "sig-final"}, + }, + ] + + parsed_chunks = [ + model_response_iterator.chunk_parser(chunk=chunk) for chunk in chunks + ] + reasoning_content = "".join( + getattr(chunk.choices[0].delta, "reasoning_content", None) or "" + for chunk in parsed_chunks + ) + thinking_blocks = tuple( + block + for chunk in parsed_chunks + for block in (getattr(chunk.choices[0].delta, "thinking_blocks", None) or []) + ) + expected_thinking_block = { + "type": "thinking", + "thinking": "Step 1. Step 2.", + "signature": "sig-final", + } + + assert reasoning_content == "Step 1. Step 2." + assert thinking_blocks == (expected_thinking_block,) + assert parsed_chunks[-1].choices[0].delta.provider_specific_fields == { + "thinking_blocks": [expected_thinking_block] + } + + +def test_streaming_unsigned_thinking_deltas_keep_reasoning_content(): + model_response_iterator = ModelResponseIterator( + streaming_response=MagicMock(), sync_stream=True, json_mode=False + ) + chunks = [ + { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "thinking", "thinking": ""}, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": "Step 1. "}, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": "Step 2."}, + }, + {"type": "content_block_stop", "index": 0}, + ] + + parsed_chunks = [ + model_response_iterator.chunk_parser(chunk=chunk) for chunk in chunks + ] + reasoning_content = "".join( + getattr(chunk.choices[0].delta, "reasoning_content", None) or "" + for chunk in parsed_chunks + ) + thinking_blocks = tuple( + block + for chunk in parsed_chunks + for block in (getattr(chunk.choices[0].delta, "thinking_blocks", None) or []) + ) + + assert reasoning_content == "Step 1. Step 2." + assert thinking_blocks == () + + +def test_streaming_truncated_thinking_deltas_keep_reasoning_content(): + model_response_iterator = ModelResponseIterator( + streaming_response=MagicMock(), sync_stream=True, json_mode=False + ) + chunks = [ + { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "thinking", "thinking": ""}, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": "Step 1. "}, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": "Step 2."}, + }, + ] + + parsed_chunks = [ + model_response_iterator.chunk_parser(chunk=chunk) for chunk in chunks + ] + reasoning_content = "".join( + getattr(chunk.choices[0].delta, "reasoning_content", None) or "" + for chunk in parsed_chunks + ) + thinking_blocks = tuple( + block + for chunk in parsed_chunks + for block in (getattr(chunk.choices[0].delta, "thinking_blocks", None) or []) + ) + + assert reasoning_content == "Step 1. Step 2." + assert thinking_blocks == () + + def test_handle_json_mode_chunk_response_format_tool(): model_response_iterator = ModelResponseIterator( streaming_response=MagicMock(), sync_stream=True, json_mode=True From 7ee492f749f2a573bacbded0c060b0ea719fd431 Mon Sep 17 00:00:00 2001 From: Rick <26716961+Bytechoreographer@users.noreply.github.com> Date: Wed, 24 Jun 2026 19:05:34 +0800 Subject: [PATCH 24/30] feat(proxy): read cold-storage prompts back in the logs detail view (#30364) * feat(proxy): read cold-storage prompts back in the logs detail view When a deployment offloads prompts and responses to cold storage instead of Postgres, the spend-log row holds only "{}" placeholders plus a metadata.cold_storage_object_key pointer, so the UI logs detail drawer showed nothing. The detail endpoint only read the placeholder columns and never fetched the object back. Resolve the payload per row based on actual content, not a config flag: if Postgres has content, return it; otherwise read the exact stored object key and fetch from the configured cold storage backend through ColdStorageHandler. Reading the persisted key is a single GET. The key embeds a microsecond timestamp that cannot be reconstructed from the millisecond-precision startTime column, and listing the day's prefix to match on request_id would be too expensive for this per-open path. Also teach the detail drawer's pretty-view parser to accept a bare messages array. The cold storage payload carries the prompt as a top-level messages list with no proxy_server_request, so without this the output rendered while the input stayed blank. ColdStorageHandler gains an optional injected logger so the resolver can be unit tested without monkeypatching. Postgres-stored prompts are unaffected: the fast path returns the existing columns and the request-body object still renders the same way. * Update litellm/proxy/spend_tracking/spend_management_endpoints.py Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> * test(proxy): cover ColdStorageHandler resolution paths and cold-storage fetch failure Add unit tests for ColdStorageHandler (injected logger, graceful None when no logger is configured, and resolution of a configured logger from the callback registry) and a regression test asserting a cold storage backend exception degrades to the Postgres values instead of surfacing a 500. --------- Co-authored-by: Bytechoreographer Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> --- .../spend_tracking/cold_storage_handler.py | 37 ++- .../spend_management_endpoints.py | 121 ++++++- .../test_spend_management_endpoints.py | 295 ++++++++++++++++++ .../PrettyMessagesView.test.tsx | 11 + .../LogDetailsDrawer/prettyMessagesUtils.ts | 24 +- 5 files changed, 447 insertions(+), 41 deletions(-) diff --git a/litellm/proxy/spend_tracking/cold_storage_handler.py b/litellm/proxy/spend_tracking/cold_storage_handler.py index 57c41bafccd..3974d4df618 100644 --- a/litellm/proxy/spend_tracking/cold_storage_handler.py +++ b/litellm/proxy/spend_tracking/cold_storage_handler.py @@ -16,8 +16,14 @@ class ColdStorageHandler: This class is responsible for handling Getting/Setting the proxy server request from cold storage. It allows fetching a dict of the proxy server request from s3 or GCS bucket. + + The cold storage logger can be injected for testing; when omitted it is + resolved from the configured ``litellm.cold_storage_custom_logger``. """ + def __init__(self, cold_storage_logger: Optional[CustomLogger] = None): + self._injected_cold_storage_logger = cold_storage_logger + async def get_proxy_server_request_from_cold_storage_with_object_key( self, object_key: str, @@ -31,33 +37,26 @@ class ColdStorageHandler: Returns: Optional[dict]: The proxy server request dict or None if not found """ - - # select the custom logger to use for cold storage - custom_logger_name: Optional[_custom_logger_compatible_callbacks_literal] = ( - self._select_custom_logger_for_cold_storage() + custom_logger = ( + self._injected_cold_storage_logger or self._resolve_cold_storage_logger() ) - - # if no custom logger name is configured, return None - if custom_logger_name is None: + if custom_logger is None: return None - # get the active/initialized custom logger - custom_logger: Optional[CustomLogger] = ( + return await custom_logger.get_proxy_server_request_from_cold_storage_with_object_key( + object_key=object_key, + ) + + def _resolve_cold_storage_logger(self) -> Optional[CustomLogger]: + custom_logger_name = self._select_custom_logger_for_cold_storage() + if custom_logger_name is None: + return None + return ( litellm.logging_callback_manager.get_active_custom_logger_for_callback_name( custom_logger_name ) ) - # if no custom logger is found, return None - if custom_logger is None: - return None - - proxy_server_request = await custom_logger.get_proxy_server_request_from_cold_storage_with_object_key( - object_key=object_key, - ) - - return proxy_server_request - def _select_custom_logger_for_cold_storage( self, ) -> Optional[_custom_logger_compatible_callbacks_literal]: diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 0ba77dcd2f0..48f12d44370 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -3,7 +3,17 @@ import collections import json import os from datetime import datetime, timedelta, timezone -from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional +from typing import ( + TYPE_CHECKING, + Any, + Dict, + List, + Literal, + Mapping, + NamedTuple, + Optional, + Union, +) import fastapi from fastapi import APIRouter, Depends, HTTPException, Request, status @@ -29,6 +39,7 @@ from litellm.repositories.verification_token_repository import ( if TYPE_CHECKING: from litellm.proxy.proxy_server import PrismaClient + from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler else: PrismaClient = Any @@ -2175,6 +2186,89 @@ async def ui_view_spend_logs( raise handle_exception_on_proxy(e) +class RequestResponsePayload(NamedTuple): + messages: Optional[Union[str, list, dict]] + response: Optional[Union[str, list, dict]] + proxy_server_request: Optional[Union[str, dict]] + + +_EMPTY_SPEND_LOG_VALUES = frozenset({"", "{}", "[]", "null"}) + + +def _spend_log_field_has_content(value: Optional[Union[str, list, dict]]) -> bool: + if value is None: + return False + if isinstance(value, str): + return value.strip() not in _EMPTY_SPEND_LOG_VALUES + if isinstance(value, (list, dict)): + return len(value) > 0 + return True + + +def _cold_storage_object_key_from_metadata( + metadata: Optional[Union[str, dict]], +) -> Optional[str]: + if isinstance(metadata, str): + try: + metadata = json.loads(metadata) + except (json.JSONDecodeError, TypeError): + return None + if not isinstance(metadata, dict): + return None + object_key = metadata.get("cold_storage_object_key") + return object_key if isinstance(object_key, str) and object_key else None + + +async def _resolve_request_response_payload( + row: Mapping[str, Any], + cold_storage_handler: "ColdStorageHandler", +) -> RequestResponsePayload: + """ + Decide where the prompt/response come from for a single spend-log row. + + PG holds the content when ``store_prompts_in_spend_logs`` is on; otherwise it + holds ``"{}"`` placeholders and the real payload lives in cold storage keyed + by ``metadata.cold_storage_object_key``. The choice is made on actual row + content, not config flags, so historical and mixed-storage rows both resolve + correctly. + """ + messages = row.get("messages") + response = row.get("response") + proxy_server_request = row.get("proxy_server_request") + + pg_payload = RequestResponsePayload(messages, response, proxy_server_request) + if ( + _spend_log_field_has_content(messages) + or _spend_log_field_has_content(response) + or _spend_log_field_has_content(proxy_server_request) + ): + return pg_payload + + object_key = _cold_storage_object_key_from_metadata(row.get("metadata")) + if object_key is None: + return pg_payload + + try: + payload = await cold_storage_handler.get_proxy_server_request_from_cold_storage_with_object_key( + object_key=object_key + ) + except Exception: + verbose_proxy_logger.warning( + "Failed to fetch cold storage payload for key %s; falling back to DB values", + object_key, + exc_info=True, + ) + return pg_payload + if payload is None: + return pg_payload + + return RequestResponsePayload( + messages=payload.get("messages"), + response=payload.get("response"), + proxy_server_request=payload.get("proxy_server_request"), + ) + + @router.get( "/spend/logs/ui/{request_id}", tags=["Budget & Spend Tracking"], @@ -2241,26 +2335,27 @@ async def ui_view_request_response_for_request_id( if payload is not None: return payload - # Fallback: fetch heavy columns directly from the database. - # The list endpoint (/spend/logs/ui) intentionally excludes messages, - # response, and proxy_server_request for performance. When no custom - # logger (S3, GCS, etc.) is configured, we still need to serve these - # fields from the DB for the detail/drawer view. + # Fallback: the list endpoint omits the heavy columns for performance, so + # serve them here. When prompts were offloaded to cold storage the DB holds + # only placeholders, so _resolve_request_response_payload fetches the real + # payload from the configured cold storage backend by object key. if prisma_client is not None: + from litellm.proxy.spend_tracking.cold_storage_handler import ( + ColdStorageHandler, + ) + sql_query = """ - SELECT messages, response, proxy_server_request + SELECT messages, response, proxy_server_request, metadata FROM "LiteLLM_SpendLogs" WHERE request_id = $1 LIMIT 1 """ db_result = await prisma_client.db.query_raw(sql_query, request_id) if db_result and len(db_result) > 0: - row = db_result[0] - return { - "messages": row.get("messages"), - "response": row.get("response"), - "proxy_server_request": row.get("proxy_server_request"), - } + resolved = await _resolve_request_response_payload( + db_result[0], cold_storage_handler=ColdStorageHandler() + ) + return resolved._asdict() return None diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index 0b583129591..b9716c22cee 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -3786,3 +3786,298 @@ async def test_ui_view_spend_logs_metadata_invalid_json_falls_back_to_empty_dict assert body["data"][0]["metadata"] == {} finally: app.dependency_overrides.pop(ps.user_api_key_auth, None) + + +class _FakeColdStorageLogger: + """Injectable cold storage logger that records the object key it was asked for.""" + + def __init__(self, payload): + self._payload = payload + self.requested_object_keys = [] + + async def get_proxy_server_request_from_cold_storage_with_object_key( + self, object_key + ): + self.requested_object_keys.append(object_key) + return self._payload + + +def _cold_storage_handler(payload): + from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler + + logger = _FakeColdStorageLogger(payload) + return ColdStorageHandler(cold_storage_logger=logger), logger + + +@pytest.mark.parametrize( + "value, expected", + [ + (None, False), + ("", False), + (" ", False), + ("{}", False), + ("[]", False), + ("null", False), + ('{"a": 1}', True), + ({}, False), + ({"a": 1}, True), + ([], False), + ([1], True), + (5, True), + ], +) +def test_spend_log_field_has_content(value, expected): + assert spend_management_endpoints._spend_log_field_has_content(value) is expected + + +@pytest.mark.parametrize( + "metadata, expected", + [ + (None, None), + ("{}", None), + ("not-json", None), + ({"cold_storage_object_key": ""}, None), + ({"cold_storage_object_key": "k/req-1.json"}, "k/req-1.json"), + ('{"cold_storage_object_key": "k/req-2.json"}', "k/req-2.json"), + ], +) +def test_cold_storage_object_key_from_metadata(metadata, expected): + assert ( + spend_management_endpoints._cold_storage_object_key_from_metadata(metadata) + == expected + ) + + +@pytest.mark.asyncio +async def test_resolve_payload_prefers_pg_and_skips_cold_storage(): + handler, logger = _cold_storage_handler({"messages": "X", "response": "Y"}) + row = { + "messages": "{}", + "response": '{"choices": [{"message": {"content": "hi"}}]}', + "proxy_server_request": "{}", + "metadata": {"cold_storage_object_key": "k/req.json"}, + } + + resolved = await spend_management_endpoints._resolve_request_response_payload( + row, cold_storage_handler=handler + ) + + assert resolved.response == '{"choices": [{"message": {"content": "hi"}}]}' + assert logger.requested_object_keys == [] + + +@pytest.mark.asyncio +async def test_resolve_payload_fetches_from_cold_storage_when_pg_empty(): + cold_payload = { + "messages": [{"role": "user", "content": "what is 2+2"}], + "response": {"choices": [{"message": {"content": "4"}}]}, + "proxy_server_request": {"body": {"model": "gpt-4o-mini"}}, + } + handler, logger = _cold_storage_handler(cold_payload) + row = { + "messages": "{}", + "response": "{}", + "proxy_server_request": "{}", + "metadata": {"cold_storage_object_key": "llm-gateway/prod/req-42.json"}, + } + + resolved = await spend_management_endpoints._resolve_request_response_payload( + row, cold_storage_handler=handler + ) + + assert logger.requested_object_keys == ["llm-gateway/prod/req-42.json"] + assert resolved.messages == cold_payload["messages"] + assert resolved.response == cold_payload["response"] + assert resolved.proxy_server_request == cold_payload["proxy_server_request"] + + +@pytest.mark.asyncio +async def test_resolve_payload_metadata_as_json_string(): + cold_payload = {"messages": "in", "response": "out", "proxy_server_request": None} + handler, logger = _cold_storage_handler(cold_payload) + row = { + "messages": "{}", + "response": "{}", + "proxy_server_request": "{}", + "metadata": json.dumps({"cold_storage_object_key": "k/str-meta.json"}), + } + + resolved = await spend_management_endpoints._resolve_request_response_payload( + row, cold_storage_handler=handler + ) + + assert logger.requested_object_keys == ["k/str-meta.json"] + assert resolved.response == "out" + + +@pytest.mark.asyncio +async def test_resolve_payload_no_object_key_returns_empty_without_fetch(): + handler, logger = _cold_storage_handler({"messages": "should-not-be-used"}) + row = { + "messages": "{}", + "response": "{}", + "proxy_server_request": "{}", + "metadata": {}, + } + + resolved = await spend_management_endpoints._resolve_request_response_payload( + row, cold_storage_handler=handler + ) + + assert logger.requested_object_keys == [] + assert resolved == spend_management_endpoints.RequestResponsePayload( + "{}", "{}", "{}" + ) + + +@pytest.mark.asyncio +async def test_resolve_payload_cold_storage_miss_falls_back_to_pg_values(): + handler, logger = _cold_storage_handler(None) + row = { + "messages": "{}", + "response": "{}", + "proxy_server_request": "{}", + "metadata": {"cold_storage_object_key": "k/missing.json"}, + } + + resolved = await spend_management_endpoints._resolve_request_response_payload( + row, cold_storage_handler=handler + ) + + assert logger.requested_object_keys == ["k/missing.json"] + assert resolved == spend_management_endpoints.RequestResponsePayload( + "{}", "{}", "{}" + ) + + +@pytest.mark.asyncio +async def test_resolve_payload_cold_storage_exception_falls_back_to_pg_values(): + """A backend error during fetch degrades to PG values instead of bubbling a 500.""" + + class _RaisingLogger: + async def get_proxy_server_request_from_cold_storage_with_object_key( + self, object_key + ): + raise RuntimeError("cold storage backend unavailable") + + from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler + + handler = ColdStorageHandler(cold_storage_logger=_RaisingLogger()) + row = { + "messages": "{}", + "response": "{}", + "proxy_server_request": "{}", + "metadata": {"cold_storage_object_key": "k/boom.json"}, + } + + resolved = await spend_management_endpoints._resolve_request_response_payload( + row, cold_storage_handler=handler + ) + + assert resolved == spend_management_endpoints.RequestResponsePayload( + "{}", "{}", "{}" + ) + + +@pytest.mark.asyncio +async def test_cold_storage_handler_uses_injected_logger(): + from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler + + logger = _FakeColdStorageLogger({"messages": "in", "response": "out"}) + handler = ColdStorageHandler(cold_storage_logger=logger) + + result = await handler.get_proxy_server_request_from_cold_storage_with_object_key( + object_key="k/req.json" + ) + + assert result == {"messages": "in", "response": "out"} + assert logger.requested_object_keys == ["k/req.json"] + + +@pytest.mark.asyncio +async def test_cold_storage_handler_returns_none_when_no_logger_configured(monkeypatch): + from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler + + monkeypatch.setattr(litellm, "cold_storage_custom_logger", None, raising=False) + handler = ColdStorageHandler() + + result = await handler.get_proxy_server_request_from_cold_storage_with_object_key( + object_key="k/req.json" + ) + + assert result is None + + +@pytest.mark.asyncio +async def test_cold_storage_handler_resolves_configured_logger_from_registry(monkeypatch): + from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler + + logger = _FakeColdStorageLogger({"messages": "from-registry"}) + monkeypatch.setattr(litellm, "cold_storage_custom_logger", "s3_v2", raising=False) + monkeypatch.setattr( + litellm.logging_callback_manager, + "get_active_custom_logger_for_callback_name", + lambda name: logger if name == "s3_v2" else None, + ) + handler = ColdStorageHandler() + + result = await handler.get_proxy_server_request_from_cold_storage_with_object_key( + object_key="k/req.json" + ) + + assert result == {"messages": "from-registry"} + assert logger.requested_object_keys == ["k/req.json"] + + +def test_ui_view_request_response_reads_from_cold_storage(client, monkeypatch): + """End-to-end: a placeholder row with a cold_storage_object_key is served from + cold storage through the detail endpoint.""" + from types import SimpleNamespace + + placeholder_row = { + "messages": "{}", + "response": "{}", + "proxy_server_request": "{}", + "metadata": {"cold_storage_object_key": "k/cold.json"}, + } + + async def _query_raw(_sql, *_args): + return [placeholder_row] + + fake_prisma = SimpleNamespace(db=SimpleNamespace(query_raw=_query_raw)) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", fake_prisma) + + cold_logger = _FakeColdStorageLogger( + { + "messages": [{"role": "user", "content": "hi"}], + "response": {"choices": [{"message": {"content": "hello"}}]}, + "proxy_server_request": None, + } + ) + monkeypatch.setattr(litellm, "cold_storage_custom_logger", "s3_v2", raising=False) + monkeypatch.setattr( + litellm.logging_callback_manager, + "get_active_additional_logging_utils_from_custom_logger", + lambda: [], + ) + monkeypatch.setattr( + litellm.logging_callback_manager, + "get_active_custom_logger_for_callback_name", + lambda name: cold_logger if name == "s3_v2" else None, + ) + + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="admin_1" + ) + try: + response = client.get( + "/spend/logs/ui/req-cold", + headers={"Authorization": "Bearer sk-test"}, + ) + assert response.status_code == 200 + body = response.json() + assert body["messages"] == [{"role": "user", "content": "hi"}] + assert body["response"] == {"choices": [{"message": {"content": "hello"}}]} + assert cold_logger.requested_object_keys == ["k/cold.json"] + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/PrettyMessagesView.test.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/PrettyMessagesView.test.tsx index b73dcafcdc3..e7295ed7a72 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/PrettyMessagesView.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/PrettyMessagesView.test.tsx @@ -27,6 +27,17 @@ describe("PrettyMessagesView", () => { expect(screen.getByText("Hi there!")).toBeInTheDocument(); }); + it("renders input when request is a bare messages array (cold storage payload)", () => { + const request = [{ role: "user", content: "Write me a poem" }]; + const response = { + choices: [{ message: { role: "assistant", content: "A quiet moment." } }], + }; + + render(); + expect(screen.getByText("Write me a poem")).toBeInTheDocument(); + expect(screen.getByText("A quiet moment.")).toBeInTheDocument(); + }); + it("should render the realtime pretty view for realtime API responses", () => { const request = {}; const response = { diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/prettyMessagesUtils.ts b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/prettyMessagesUtils.ts index 32ae294b1ee..09b8f551c1d 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/prettyMessagesUtils.ts +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/prettyMessagesUtils.ts @@ -39,18 +39,24 @@ export const ROLE_STYLES: Record = { * Parse request messages and response message from log data */ export const parseMessages = (request: any, response: any): ParsedMessages => { - // Parse request messages + // Parse request messages. `request` is either the raw request body + // ({ messages: [...] }) or, when prompts come from cold storage, the bare + // messages array itself. const requestMessages: ParsedMessage[] = []; - if (request?.messages && Array.isArray(request.messages)) { - request.messages.forEach((msg: any) => { - requestMessages.push({ - role: msg.role || "user", - content: parseMessageContent(msg.content), - toolCallId: msg.tool_call_id, - }); + const requestMessageList = Array.isArray(request) + ? request + : Array.isArray(request?.messages) + ? request.messages + : []; + + requestMessageList.forEach((msg: any) => { + requestMessages.push({ + role: msg.role || "user", + content: parseMessageContent(msg.content), + toolCallId: msg.tool_call_id, }); - } + }); // Parse response message let responseMessage: ParsedMessage | null = null; From 1eb6bdde9c46aa4417e4ffb049ac777ea4cd1ec6 Mon Sep 17 00:00:00 2001 From: Praveen Ghuge Date: Wed, 24 Jun 2026 16:36:52 +0530 Subject: [PATCH 25/30] fix(mavvrik): advance metricsMarker after upload; fix scheduler startup (#31068) * fix(mavvrik): advance metricsMarker after upload + fix scheduler startup Two bugs fixed: 1. deliver() never called PATCH /metrics/agent/ai/{connectionId} after a successful GCS upload, so metricsMarker stayed at 0 and every daily run re-exported the same dates in an infinite catch-up loop. Fix: add _update_metrics_marker(date_epoch) called at the end of deliver() after _upload_to_gcs() succeeds. A 4xx warns but does not raise (the GCS file is already committed). A 410 raises consistent with the rest of the destination. 2. init_mavvrik_focus_background_job runs at proxy startup before any LLM call has triggered lazy instantiation of MavvrikFocusLogger, so it found no logger instance and silently skipped registering the daily export job. Fix: if no instance is found but "mavvrik" is in litellm.callbacks, call _init_custom_logger_compatible_class to force instantiation before the APScheduler job is registered. * fix(mavvrik): catch up from earliest window when metricsMarker=0 When the connector is freshly registered, metricsMarker=0 parses to None. The catch-up block was guarded by `if last_ingested and ...` which skipped it entirely for None, so only yesterday was exported instead of the full _MAX_CATCHUP_DAYS window. Fix: treat None as being _MAX_CATCHUP_DAYS behind (start from earliest_catchup). The existing > 7 day warning only fires for non-None markers that are old. * fix(mavvrik): use now as end_time for yesterday's export window LiteLLM_DailyUserSpend rows for a given date get their updated_at bumped by the spend flush job throughout the next morning. The core database query filters on updated_at, so capping end_time at midnight (yesterday + 1 day) missed any spend rows flushed after midnight. Fix: pass now (cron fire time) as end_time for the daily "yesterday" window so all fully-settled rows are captured regardless of when the flush job ran. Verified: claude-3-5-sonnet BilledCost went from 0.0 to ~$2.40 per row in the exported FOCUS CSV. * fix(mavvrik): also use now as end_time for catch-up windows * fix(mavvrik_focus): pass required args to _init_custom_logger_compatible_class Calling it with only logging_integration raised TypeError at proxy startup because internal_usage_cache and llm_router have no defaults. Also fix test name to reflect the actual status code (5xx not 4xx) used in the mock. * ci: retrigger CI run --- .../focus/destinations/mavvrik_destination.py | 36 ++++++-- .../mavvrik_focus/mavvrik_focus_logger.py | 51 ++++++++--- .../focus/test_mavvrik_destination.py | 86 ++++++++++++++++--- 3 files changed, 143 insertions(+), 30 deletions(-) diff --git a/litellm/integrations/focus/destinations/mavvrik_destination.py b/litellm/integrations/focus/destinations/mavvrik_destination.py index 1e3c98b9a70..659f608a3e1 100644 --- a/litellm/integrations/focus/destinations/mavvrik_destination.py +++ b/litellm/integrations/focus/destinations/mavvrik_destination.py @@ -3,6 +3,7 @@ Flow: 1. GET /metrics/agent/ai/{connection_id}/upload-url → GCS signed URL 2. PUT with CSV content + 3. PATCH /metrics/agent/ai/{connection_id} → advance metricsMarker """ from __future__ import annotations @@ -127,8 +128,6 @@ class FocusMavvrikDestination(FocusDestination): timeout=30.0, ) if resp.status_code == 410: - # Connector has been disconnected in Mavvrik — reset flag so next - # delivery attempt re-registers after it becomes active again. self._registered = False raise RuntimeError( "Mavvrik FOCUS destination: connector is disconnected (410). " @@ -273,14 +272,35 @@ class FocusMavvrikDestination(FocusDestination): pass raise + async def _update_metrics_marker(self, date_epoch: int) -> None: + """PATCH agent endpoint to advance metricsMarker after a successful upload.""" + resp = await self._http.client.request( + method="PATCH", + url=self._agent_url, + headers=self._auth_headers, + json={"metricsMarker": date_epoch}, + timeout=30.0, + ) + if resp.status_code == 410: + self._registered = False + raise RuntimeError( + "Mavvrik FOCUS destination: connector is disconnected (410). " + "Re-enable the connection in the Mavvrik dashboard." + ) + if resp.status_code >= 400: + verbose_logger.warning( + "Mavvrik FOCUS destination: failed to update metricsMarker (%s): %s", + resp.status_code, + resp.text[:200], + ) + return + verbose_logger.debug( + "Mavvrik FOCUS destination: metricsMarker advanced to %s", date_epoch + ) + async def get_metrics_marker(self) -> Optional[int]: """Register with Mavvrik and return the current metricsMarker. - The metricsMarker is a Unix timestamp (seconds) representing the last - date Mavvrik has successfully ingested. Called on every scheduled run - so the logger can detect and catch up any dates missed due to previous - export failures. - Always calls the Mavvrik register API — unlike deliver() which skips registration once _registered is True, catch-up requires a fresh marker value on every run. @@ -328,6 +348,7 @@ class FocusMavvrikDestination(FocusDestination): return date_str = time_window.start_time.strftime("%Y-%m-%d") + date_epoch = int(time_window.start_time.timestamp()) verbose_logger.debug( "Mavvrik FOCUS destination: uploading %d bytes for date=%s (%s)", @@ -339,6 +360,7 @@ class FocusMavvrikDestination(FocusDestination): await self._ensure_registered() signed_url = await self._get_signed_url(date_str) await self._upload_to_gcs(signed_url, content) + await self._update_metrics_marker(date_epoch) verbose_logger.debug( "Mavvrik FOCUS destination: upload complete for date=%s", date_str diff --git a/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py b/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py index 47d3e1da7bc..bbc9d1a6330 100644 --- a/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py +++ b/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py @@ -149,8 +149,8 @@ class MavvrikFocusLogger(FocusLogger): On each run: 1. Register with Mavvrik → get metricsMarker (last successfully ingested date) - 2. If metricsMarker is behind yesterday, catch up missed dates (capped at - _MAX_CATCHUP_DAYS to avoid runaway loops on long outages) + 2. If metricsMarker is behind yesterday (or 0/None for a fresh connector), + catch up missed dates (capped at _MAX_CATCHUP_DAYS) 3. Export yesterday (today's daily window) This ensures a failed export on day N is automatically retried on day N+1 @@ -177,13 +177,21 @@ class MavvrikFocusLogger(FocusLogger): last_ingested = _parse_metrics_marker(marker) - # Catch up missed dates, capped at _MAX_CATCHUP_DAYS - if last_ingested and last_ingested < yesterday: - # Never go further back than _MAX_CATCHUP_DAYS from yesterday - earliest_catchup = yesterday - timedelta(days=self._MAX_CATCHUP_DAYS - 1) - catch_up_date = max(last_ingested + timedelta(days=1), earliest_catchup) + # Catch up missed dates, capped at _MAX_CATCHUP_DAYS. + # last_ingested=None means metricsMarker=0 (fresh connector, never ingested) — + # treat the same as being _MAX_CATCHUP_DAYS behind so we export all available history. + earliest_catchup = yesterday - timedelta(days=self._MAX_CATCHUP_DAYS - 1) + if last_ingested is None or last_ingested < yesterday: + catch_up_date = ( + earliest_catchup + if last_ingested is None + else max(last_ingested + timedelta(days=1), earliest_catchup) + ) - if last_ingested + timedelta(days=1) < earliest_catchup: + if ( + last_ingested is not None + and last_ingested + timedelta(days=1) < earliest_catchup + ): verbose_proxy_logger.warning( "Mavvrik FOCUS export: metricsMarker is more than %d days behind " "(%s). Catching up from %s only; earlier data will not be re-exported.", @@ -197,18 +205,24 @@ class MavvrikFocusLogger(FocusLogger): "Mavvrik FOCUS export: catching up missed date %s", catch_up_date.date(), ) + # Use now as end_time for catch-up windows too — rows for old dates + # may have been flushed to DB well after their calendar day ended. + catch_up_end = min(catch_up_date + timedelta(days=1), now) window = FocusTimeWindow( start_time=catch_up_date, - end_time=catch_up_date + timedelta(days=1), + end_time=catch_up_end, frequency="daily", ) await self._export_window(window=window, limit=None) catch_up_date += timedelta(days=1) - # Export yesterday's window (the normal daily run) + # Export yesterday's window (the normal daily run). + # Use `now` as end_time so spend rows flushed after midnight are included. + # LiteLLM's DailyUserSpend rows for a given date keep getting updated_at + # bumped as the flush job runs; capping at midnight would miss those updates. window = FocusTimeWindow( start_time=yesterday, - end_time=yesterday + timedelta(days=1), + end_time=now, frequency="daily", ) await self._export_window(window=window, limit=None) @@ -253,6 +267,21 @@ class MavvrikFocusLogger(FocusLogger): ) if type(cb) is MavvrikFocusLogger ] + if not loggers and "mavvrik" in litellm.callbacks: + # The logger is registered as the string "mavvrik" but hasn't been + # instantiated yet (lazy init happens on first LLM call). Force it now + # so the scheduler can register the daily export job at startup. + from litellm.litellm_core_utils.litellm_logging import ( # noqa: PLC0415 + _init_custom_logger_compatible_class, + ) + + instance = _init_custom_logger_compatible_class( + logging_integration="mavvrik", + internal_usage_cache=None, + llm_router=None, + ) + if isinstance(instance, MavvrikFocusLogger): + loggers = [instance] if not loggers: verbose_proxy_logger.debug( "No MavvrikFocusLogger registered; skipping scheduler" diff --git a/tests/test_litellm/integrations/focus/test_mavvrik_destination.py b/tests/test_litellm/integrations/focus/test_mavvrik_destination.py index 797238ae238..3a23dc4ffb2 100644 --- a/tests/test_litellm/integrations/focus/test_mavvrik_destination.py +++ b/tests/test_litellm/integrations/focus/test_mavvrik_destination.py @@ -34,6 +34,12 @@ def _dest(**overrides) -> FocusMavvrikDestination: return FocusMavvrikDestination(prefix="mavvrik_focus_exports", config=config) +def _patch_resp(status: int = 204) -> MagicMock: + r = MagicMock() + r.status_code = status + return r + + def test_missing_api_key_raises(): with pytest.raises(ValueError, match="MAVVRIK_API_KEY"): FocusMavvrikDestination( @@ -127,6 +133,8 @@ async def test_large_content_uploads_in_multiple_chunks(): chunk2_resp = MagicMock() chunk2_resp.status_code = 200 + patch_resp = _patch_resp(204) + mock_http = MagicMock() mock_http.client = MagicMock() mock_http.client.request = AsyncMock( @@ -136,6 +144,7 @@ async def test_large_content_uploads_in_multiple_chunks(): init_resp, chunk1_resp, chunk2_resp, + patch_resp, ] ) dest._http = mock_http @@ -152,15 +161,20 @@ async def test_large_content_uploads_in_multiple_chunks(): filename="usage.csv", ) - # register + get_signed_url + init + 2 chunk PUTs = 5 calls - assert mock_http.client.request.call_count == 5 + # register + get_signed_url + init + 2 chunk PUTs + PATCH = 6 calls + assert mock_http.client.request.call_count == 6 - # Check Content-Range headers - put_calls = mock_http.client.request.call_args_list[3:] + # Check Content-Range headers on the chunk PUTs (calls 3 and 4) + put_calls = mock_http.client.request.call_args_list[3:5] assert "bytes" in put_calls[0].kwargs["headers"]["Content-Range"] assert "/*" in put_calls[0].kwargs["headers"]["Content-Range"] # intermediate assert "/*" not in put_calls[1].kwargs["headers"]["Content-Range"] # final + # Verify the PATCH call advanced metricsMarker + patch_call = mock_http.client.request.call_args_list[5] + assert patch_call.kwargs["method"] == "PATCH" + assert "metricsMarker" in patch_call.kwargs["json"] + @pytest.mark.asyncio async def test_deliver_calls_register_get_url_and_upload(): @@ -180,12 +194,14 @@ async def test_deliver_calls_register_get_url_and_upload(): upload_resp = MagicMock() upload_resp.status_code = 200 + patch_resp = _patch_resp(204) + mock_http = MagicMock() mock_http.client = MagicMock() - # All 4 calls go through self._http.client.request: - # 1. register, 2. get_signed_url, 3. GCS session init POST, 4. GCS PUT + # All 5 calls go through self._http.client.request: + # 1. register, 2. get_signed_url, 3. GCS session init POST, 4. GCS PUT, 5. PATCH marker mock_http.client.request = AsyncMock( - side_effect=[register_resp, signed_url_resp, init_resp, upload_resp] + side_effect=[register_resp, signed_url_resp, init_resp, upload_resp, patch_resp] ) dest._http = mock_http @@ -196,10 +212,14 @@ async def test_deliver_calls_register_get_url_and_upload(): ) assert dest._registered is True - assert mock_http.client.request.call_count == 4 + assert mock_http.client.request.call_count == 5 # Verify Content-Range header was set on the PUT put_call = mock_http.client.request.call_args_list[3] assert "Content-Range" in put_call.kwargs["headers"] + # Verify PATCH was called last with metricsMarker + patch_call = mock_http.client.request.call_args_list[4] + assert patch_call.kwargs["method"] == "PATCH" + assert "metricsMarker" in patch_call.kwargs["json"] @pytest.mark.asyncio @@ -224,17 +244,19 @@ async def test_register_called_only_once_across_multiple_deliveries(): mock_http = MagicMock() mock_http.client = MagicMock() - # First delivery: register, get_signed_url, GCS init, GCS PUT - # Second delivery: get_signed_url, GCS init, GCS PUT (register skipped) + # First delivery: register, get_signed_url, GCS init, GCS PUT, PATCH + # Second delivery: get_signed_url, GCS init, GCS PUT, PATCH (register skipped) mock_http.client.request = AsyncMock( side_effect=[ register_resp, _signed_url_resp(), init_resp, upload_resp, + _patch_resp(204), _signed_url_resp(), init_resp, upload_resp, + _patch_resp(204), ] ) dest._http = mock_http @@ -243,8 +265,8 @@ async def test_register_called_only_once_across_multiple_deliveries(): await dest.deliver(content=b"header\nrow1\n", time_window=window, filename="1.csv") await dest.deliver(content=b"header\nrow2\n", time_window=window, filename="2.csv") - # 7 total: register(1) + [get_url+init+put](2) × 2 deliveries - assert mock_http.client.request.call_count == 7 + # 9 total: register(1) + [get_url+init+put+patch](4) × 2 deliveries + assert mock_http.client.request.call_count == 9 # First call was register first_call = mock_http.client.request.call_args_list[0] assert first_call.kwargs["method"] == "POST" @@ -749,3 +771,43 @@ async def test_gcs_session_cancelled_on_chunk_failure(): delete_call = calls[4] assert delete_call.kwargs["method"] == "DELETE" assert "storage.googleapis.com/session" in delete_call.kwargs["url"] + + +@pytest.mark.asyncio +async def test_update_metrics_marker_warns_on_non_410_error(): + """_update_metrics_marker must log a warning on any >=400 (non-410) status but not raise.""" + dest = _dest() + + fail_resp = MagicMock() + fail_resp.status_code = 500 + fail_resp.text = "Internal Server Error" + + mock_http = MagicMock() + mock_http.client = MagicMock() + mock_http.client.request = AsyncMock(return_value=fail_resp) + dest._http = mock_http + + # Must not raise — warning only + await dest._update_metrics_marker(1234567890) + assert mock_http.client.request.call_count == 1 + + +@pytest.mark.asyncio +async def test_update_metrics_marker_raises_on_410(): + """_update_metrics_marker must raise RuntimeError and reset _registered on 410.""" + dest = _dest() + dest._registered = True + + resp_410 = MagicMock() + resp_410.status_code = 410 + resp_410.text = "Gone" + + mock_http = MagicMock() + mock_http.client = MagicMock() + mock_http.client.request = AsyncMock(return_value=resp_410) + dest._http = mock_http + + with pytest.raises(RuntimeError, match="disconnected"): + await dest._update_metrics_marker(1234567890) + + assert dest._registered is False From e1187c0462a1c456c1eb74f09ac9d907fe0ce64a Mon Sep 17 00:00:00 2001 From: Jim Smith Date: Wed, 24 Jun 2026 07:15:38 -0400 Subject: [PATCH 26/30] feat: pass through optional `instruction` field in the rerank API (vLLM/Qwen3-Reranker) (#30757) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Add optional `instruction` passthrough to the rerank API vLLM's /v1/rerank and /v1/score accept an optional top-level `instruction` field (folded into the model's chat_template_kwargs and consumed by the chat template — e.g. Qwen3-Reranker). LiteLLM's managed rerank route silently dropped it: RerankRequest / OptionalRerankParams had no such field, so the outgoing body was rebuilt without it. Thread an opt-in `instruction: Optional[str]` through rerank()/arerank(), get_optional_rerank_params, and the hosted_vllm transformation into the request body, only when non-None. When callers omit it, model_dump(exclude_none) drops the field and the outgoing request is byte-for-byte unchanged — fully backward-compatible. (DeepInfra already forwards `instruction` via non_default_params; this formalizes the field in the shared types.) Co-Authored-By: Claude Opus 4.8 (1M context) * Address review: thread `instruction` as a typed param + cover rerank_utils Per PR review (greptile P2 + codecov): - Make `instruction` a typed, named argument on the rerank provider interface instead of recovering it from the opaque `non_default_params` blob. Adds `instruction: Optional[str] = None` to `BaseRerankConfig.map_cohere_rerank_params` and every provider override, and forwards it explicitly from `get_optional_rerank_params`. hosted_vllm now reads the named param directly. It is still also surfaced in `non_default_params` so providers that read it there (e.g. DeepInfra) keep working now that `rerank()` consumes `instruction` as a named param rather than leaving it in **kwargs. - Add get_optional_rerank_params unit tests (present + absent) to cover the previously-uncovered threading line flagged by codecov. Co-Authored-By: Claude Opus 4.8 (1M context) * fix: scan rerank `instruction` through request guardrails The rerank guardrail translation (CohereRerankHandler.process_input_messages) only scanned `query`, so the newly added `instruction` field reached the backend model unscanned. Since instruction-aware rerankers (hosted vLLM / Qwen3-Reranker) fold `instruction` into the prompt, an authenticated caller could place content there to bypass configured rerank request guardrails. Generalize the handler to scan every user-controlled text field (`query` and `instruction`) in one apply_guardrail call and write each sanitized value back by index. Query-only requests are unchanged (single-element list at index 0); non-string fields are left untouched. Adds tests covering instruction scanning, PII masking write-back, and the non-string case. Addresses the Veria AI security review on PR #30757. * test: narrow Optional results before len() to satisfy basedpyright budget The lint gate (basedpyright delta-vs-base budget) flagged one new reportArgumentType: len(result.results) where results is List[RerankResponseResult] | None. Assert results is not None first to narrow the type before len()/indexing. * fix: read rerank `instruction` from kwargs to satisfy basedpyright budget The basedpyright delta-vs-base gate flagged one new reportArgumentType: the Router forwards rerank calls via an untyped `**kwargs` unpack (`litellm.arerank(**{**data, **kwargs})`), and declaring `instruction` as a typed named param on the public `rerank`/`arerank` entrypoints made pyright check that key against `str | None`, adding an error at router.py with no real safety gain. Read `instruction` from kwargs in `rerank` instead. It remains fully typed where it matters - threaded as a typed argument through `get_optional_rerank_params` and each provider's `map_cohere_rerank_params` (the original Greptile P2 ask). Whole-repo reportArgumentType is back to the base count (net 0); rerank hosted_vllm + cohere guardrail suites pass; ruff clean. --------- Co-authored-by: Claude Opus 4.8 (1M context) --- .../llms/base_llm/rerank/transformation.py | 1 + .../rerank/guardrail_translation/handler.py | 76 +++++++++++------- litellm/llms/cohere/rerank/transformation.py | 1 + .../llms/cohere/rerank_v2/transformation.py | 1 + .../llms/dashscope/rerank/transformation.py | 1 + .../llms/deepinfra/rerank/transformation.py | 1 + .../fireworks_ai/rerank/transformation.py | 1 + .../llms/hosted_vllm/rerank/transformation.py | 25 ++++-- .../llms/huggingface/rerank/transformation.py | 1 + litellm/llms/jina_ai/rerank/transformation.py | 1 + .../llms/nvidia_nim/rerank/transformation.py | 1 + .../llms/vertex_ai/rerank/transformation.py | 1 + litellm/llms/voyage/rerank/transformation.py | 1 + litellm/llms/watsonx/rerank/transformation.py | 1 + litellm/rerank_api/main.py | 6 ++ litellm/rerank_api/rerank_utils.py | 7 ++ litellm/types/rerank.py | 5 ++ .../rerank/test_rerank_guardrail_handler.py | 70 +++++++++++++++- .../test_hosted_vllm_rerank_transformation.py | 79 +++++++++++++++++++ 19 files changed, 242 insertions(+), 38 deletions(-) diff --git a/litellm/llms/base_llm/rerank/transformation.py b/litellm/llms/base_llm/rerank/transformation.py index 166f876ba04..6603c64142b 100644 --- a/litellm/llms/base_llm/rerank/transformation.py +++ b/litellm/llms/base_llm/rerank/transformation.py @@ -85,6 +85,7 @@ class BaseRerankConfig(ABC): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: pass diff --git a/litellm/llms/cohere/rerank/guardrail_translation/handler.py b/litellm/llms/cohere/rerank/guardrail_translation/handler.py index e9a5823d2b8..0824e1cca41 100644 --- a/litellm/llms/cohere/rerank/guardrail_translation/handler.py +++ b/litellm/llms/cohere/rerank/guardrail_translation/handler.py @@ -26,11 +26,18 @@ class CohereRerankHandler(BaseTranslation): The handler specifically processes: - The 'query' parameter (string) + - The 'instruction' parameter (string), when present Note: Documents are not processed by guardrails as they are the corpus being searched, not user input. """ + # User-controlled free-text fields that reach the model and must be + # scanned. 'instruction' is folded into the prompt by instruction-aware + # rerankers (e.g. hosted vLLM / Qwen3-Reranker), so it is as sensitive as + # 'query'; omitting it would let a caller smuggle content past guardrails. + _SCANNED_FIELDS = ("query", "instruction") + async def process_input_messages( self, data: dict, @@ -38,42 +45,55 @@ class CohereRerankHandler(BaseTranslation): litellm_logging_obj: Optional[Any] = None, ) -> Any: """ - Process input query by applying guardrails. + Process input text fields ('query' and 'instruction') by applying + guardrails and writing the sanitized values back. Args: - data: Request data dictionary containing 'query' + data: Request data dictionary containing 'query' and optionally + 'instruction' guardrail_to_apply: The guardrail instance to apply Returns: - Modified data with guardrails applied to query only + Modified data with guardrails applied to query/instruction only """ - # Process query only - query = data.get("query") - if query is not None and isinstance(query, str): - inputs = GenericGuardrailAPIInputs(texts=[query]) - # Include model information if available - model = data.get("model") - if model: - inputs["model"] = model - guardrailed_inputs = await guardrail_to_apply.apply_guardrail( - inputs=inputs, - request_data=data, - input_type="request", - logging_obj=litellm_logging_obj, + # Collect every scannable text field in a stable order so the + # guardrailed results can be written back to the right key by index. + fields_to_scan = [ + (key, data[key]) + for key in self._SCANNED_FIELDS + if isinstance(data.get(key), str) + ] + if not fields_to_scan: + verbose_proxy_logger.debug( + "Rerank: No query/instruction to process or not strings" ) - guardrailed_texts = guardrailed_inputs.get("texts", []) - data["query"] = guardrailed_texts[0] if guardrailed_texts else query + return data - verbose_proxy_logger.debug( - "Rerank: Applied guardrail to query. " - "Original length: %d, New length: %d", - len(query), - len(data["query"]), - ) - else: - verbose_proxy_logger.debug( - "Rerank: No query to process or query is not a string" - ) + inputs = GenericGuardrailAPIInputs(texts=[value for _, value in fields_to_scan]) + # Include model information if available + model = data.get("model") + if model: + inputs["model"] = model + guardrailed_inputs = await guardrail_to_apply.apply_guardrail( + inputs=inputs, + request_data=data, + input_type="request", + logging_obj=litellm_logging_obj, + ) + guardrailed_texts = guardrailed_inputs.get("texts", []) + + for idx, (key, original) in enumerate(fields_to_scan): + # Defensive: only write back when the guardrail returned a value for + # this index; otherwise keep the original (never forward unscanned). + if idx < len(guardrailed_texts): + data[key] = guardrailed_texts[idx] + verbose_proxy_logger.debug( + "Rerank: Applied guardrail to %s. " + "Original length: %d, New length: %d", + key, + len(original), + len(data[key]), + ) return data diff --git a/litellm/llms/cohere/rerank/transformation.py b/litellm/llms/cohere/rerank/transformation.py index 64ae8e8ffa7..d875f420310 100644 --- a/litellm/llms/cohere/rerank/transformation.py +++ b/litellm/llms/cohere/rerank/transformation.py @@ -57,6 +57,7 @@ class CohereRerankConfig(BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: """ Map Cohere rerank params diff --git a/litellm/llms/cohere/rerank_v2/transformation.py b/litellm/llms/cohere/rerank_v2/transformation.py index 4c800d6455d..0dcb10d5664 100644 --- a/litellm/llms/cohere/rerank_v2/transformation.py +++ b/litellm/llms/cohere/rerank_v2/transformation.py @@ -49,6 +49,7 @@ class CohereRerankV2Config(CohereRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: """ Map Cohere rerank params diff --git a/litellm/llms/dashscope/rerank/transformation.py b/litellm/llms/dashscope/rerank/transformation.py index 629f3cf4af7..745e85de7e3 100644 --- a/litellm/llms/dashscope/rerank/transformation.py +++ b/litellm/llms/dashscope/rerank/transformation.py @@ -116,6 +116,7 @@ class DashScopeRerankConfig(BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: # qwen3-rerank accepts query/documents/top_n/return_documents. The # rest (rank_fields, max_*_per_doc) are silently dropped. diff --git a/litellm/llms/deepinfra/rerank/transformation.py b/litellm/llms/deepinfra/rerank/transformation.py index e4bfbcb2513..a5c36ca2e5f 100644 --- a/litellm/llms/deepinfra/rerank/transformation.py +++ b/litellm/llms/deepinfra/rerank/transformation.py @@ -104,6 +104,7 @@ class DeepinfraRerankConfig(BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: # Start with the basic parameters optional_rerank_params = {} diff --git a/litellm/llms/fireworks_ai/rerank/transformation.py b/litellm/llms/fireworks_ai/rerank/transformation.py index 4a7b64b9b77..27309780c86 100644 --- a/litellm/llms/fireworks_ai/rerank/transformation.py +++ b/litellm/llms/fireworks_ai/rerank/transformation.py @@ -67,6 +67,7 @@ class FireworksAIRerankConfig(FireworksAIMixin, BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict[str, Any]: """ Map Cohere rerank params to Fireworks AI rerank params diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py index 60b6dc7d23d..d0c96f8b420 100644 --- a/litellm/llms/hosted_vllm/rerank/transformation.py +++ b/litellm/llms/hosted_vllm/rerank/transformation.py @@ -61,6 +61,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig): "top_n", "rank_fields", "return_documents", + "instruction", ] def map_cohere_rerank_params( @@ -76,6 +77,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: """ Map parameters for Hosted VLLM rerank @@ -83,16 +85,22 @@ class HostedVLLMRerankConfig(BaseRerankConfig): if max_chunks_per_doc is not None: raise ValueError("Hosted VLLM does not support max_chunks_per_doc") - return dict( - OptionalRerankParams( - query=query, - documents=documents, - top_n=top_n, - rank_fields=rank_fields, - return_documents=return_documents, - ) + mapped_params = OptionalRerankParams( + query=query, + documents=documents, + top_n=top_n, + rank_fields=rank_fields, + return_documents=return_documents, ) + # `instruction` is a vLLM-supported passthrough (folded into the model's + # chat_template_kwargs). Only forward it when explicitly set so omitting + # it leaves the request unchanged. + if instruction is not None: + mapped_params["instruction"] = instruction + + return dict(mapped_params) + def validate_environment( self, headers: dict, @@ -135,6 +143,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig): top_n=optional_rerank_params.get("top_n", None), rank_fields=optional_rerank_params.get("rank_fields", None), return_documents=optional_rerank_params.get("return_documents", None), + instruction=optional_rerank_params.get("instruction", None), ) return rerank_request.model_dump(exclude_none=True) diff --git a/litellm/llms/huggingface/rerank/transformation.py b/litellm/llms/huggingface/rerank/transformation.py index 2c847b617ef..4e409f31ed2 100644 --- a/litellm/llms/huggingface/rerank/transformation.py +++ b/litellm/llms/huggingface/rerank/transformation.py @@ -100,6 +100,7 @@ class HuggingFaceRerankConfig(BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: optional_rerank_params = {} if non_default_params is not None: diff --git a/litellm/llms/jina_ai/rerank/transformation.py b/litellm/llms/jina_ai/rerank/transformation.py index 56be754fc34..0d48ed5edcd 100644 --- a/litellm/llms/jina_ai/rerank/transformation.py +++ b/litellm/llms/jina_ai/rerank/transformation.py @@ -45,6 +45,7 @@ class JinaAIRerankConfig(BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: optional_params = {} supported_params = self.get_supported_cohere_rerank_params(model) diff --git a/litellm/llms/nvidia_nim/rerank/transformation.py b/litellm/llms/nvidia_nim/rerank/transformation.py index fc317293acc..8eee188bf46 100644 --- a/litellm/llms/nvidia_nim/rerank/transformation.py +++ b/litellm/llms/nvidia_nim/rerank/transformation.py @@ -117,6 +117,7 @@ class NvidiaNimRerankConfig(BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: """ Map Cohere/OpenAI rerank params to Nvidia NIM format. diff --git a/litellm/llms/vertex_ai/rerank/transformation.py b/litellm/llms/vertex_ai/rerank/transformation.py index 3b84972e946..d2041009efb 100644 --- a/litellm/llms/vertex_ai/rerank/transformation.py +++ b/litellm/llms/vertex_ai/rerank/transformation.py @@ -242,6 +242,7 @@ class VertexAIRerankConfig(BaseRerankConfig, VertexBase): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: """ Map Cohere rerank params to Vertex AI format diff --git a/litellm/llms/voyage/rerank/transformation.py b/litellm/llms/voyage/rerank/transformation.py index d64450a1211..907e5b7e26b 100644 --- a/litellm/llms/voyage/rerank/transformation.py +++ b/litellm/llms/voyage/rerank/transformation.py @@ -39,6 +39,7 @@ class VoyageRerankConfig(BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: # Voyage AI uses 'top_k' instead of 'top_n' optional_params: Dict[str, Any] = {"query": query, "documents": documents} diff --git a/litellm/llms/watsonx/rerank/transformation.py b/litellm/llms/watsonx/rerank/transformation.py index 202760f68a6..a34358a6be3 100644 --- a/litellm/llms/watsonx/rerank/transformation.py +++ b/litellm/llms/watsonx/rerank/transformation.py @@ -104,6 +104,7 @@ class IBMWatsonXRerankConfig(IBMWatsonXMixin, BaseRerankConfig): return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, ) -> Dict: """ Map Cohere rerank params to IBM watsonx.ai rerank params diff --git a/litellm/rerank_api/main.py b/litellm/rerank_api/main.py index e40e12e9197..3ef74d596ad 100644 --- a/litellm/rerank_api/main.py +++ b/litellm/rerank_api/main.py @@ -103,6 +103,11 @@ def rerank( """ Reranks a list of documents based on their relevance to the query """ + # `instruction` is read from kwargs rather than declared as a named param. + # The router forwards rerank calls via an untyped `**kwargs` unpack, and a + # typed named param there would trip the basedpyright budget gate without + # adding real safety; it stays typed downstream via get_optional_rerank_params. + instruction: Optional[str] = kwargs.get("instruction", None) headers: Optional[dict] = kwargs.get("headers") # type: ignore litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None) @@ -155,6 +160,7 @@ def rerank( return_documents=return_documents, max_chunks_per_doc=max_chunks_per_doc, max_tokens_per_doc=max_tokens_per_doc, + instruction=instruction, non_default_params=kwargs, ) verbose_logger.info(f"optional_rerank_params: {optional_rerank_params}") diff --git a/litellm/rerank_api/rerank_utils.py b/litellm/rerank_api/rerank_utils.py index 38e599ef824..a8a665496fc 100644 --- a/litellm/rerank_api/rerank_utils.py +++ b/litellm/rerank_api/rerank_utils.py @@ -15,6 +15,7 @@ def get_optional_rerank_params( return_documents: Optional[bool] = True, max_chunks_per_doc: Optional[int] = None, max_tokens_per_doc: Optional[int] = None, + instruction: Optional[str] = None, non_default_params: Optional[dict] = None, ) -> Dict: all_non_default_params = non_default_params or {} @@ -30,6 +31,11 @@ def get_optional_rerank_params( all_non_default_params["max_chunks_per_doc"] = max_chunks_per_doc if max_tokens_per_doc is not None: all_non_default_params["max_tokens_per_doc"] = max_tokens_per_doc + if instruction is not None: + # Also surfaced in non_default_params so providers that read it from + # there (e.g. DeepInfra) keep working now that `rerank()` consumes + # `instruction` as a named param instead of leaving it in **kwargs. + all_non_default_params["instruction"] = instruction return rerank_provider_config.map_cohere_rerank_params( model=model, drop_params=drop_params, @@ -41,5 +47,6 @@ def get_optional_rerank_params( return_documents=return_documents, max_chunks_per_doc=max_chunks_per_doc, max_tokens_per_doc=max_tokens_per_doc, + instruction=instruction, non_default_params=all_non_default_params, ) diff --git a/litellm/types/rerank.py b/litellm/types/rerank.py index d2c252a1e92..376d6f66603 100644 --- a/litellm/types/rerank.py +++ b/litellm/types/rerank.py @@ -19,6 +19,10 @@ class RerankRequest(BaseModel): return_documents: Optional[bool] = None max_chunks_per_doc: Optional[int] = None max_tokens_per_doc: Optional[int] = None + # Optional task/query instruction passed through to providers that support it + # (e.g. hosted vLLM / Qwen3-Reranker, DeepInfra). Omitted from the outgoing + # request when None, so this is fully backward-compatible. + instruction: Optional[str] = None class OptionalRerankParams(TypedDict, total=False): @@ -29,6 +33,7 @@ class OptionalRerankParams(TypedDict, total=False): return_documents: Optional[bool] max_chunks_per_doc: Optional[int] max_tokens_per_doc: Optional[int] + instruction: Optional[str] class RerankBilledUnits(TypedDict, total=False): diff --git a/tests/test_litellm/llms/cohere/rerank/test_rerank_guardrail_handler.py b/tests/test_litellm/llms/cohere/rerank/test_rerank_guardrail_handler.py index 88072cd7760..46c37e6af6c 100644 --- a/tests/test_litellm/llms/cohere/rerank/test_rerank_guardrail_handler.py +++ b/tests/test_litellm/llms/cohere/rerank/test_rerank_guardrail_handler.py @@ -2,10 +2,8 @@ Unit tests for Cohere Rerank Guardrail Translation Handler """ -import asyncio import os import sys -from typing import List, Optional, Tuple import pytest @@ -94,6 +92,74 @@ class TestInputProcessing: "id": "doc2", } + @pytest.mark.asyncio + async def test_process_query_and_instruction(self): + """Both query and instruction are guardrailed; documents untouched""" + handler = CohereRerankHandler() + guardrail = MockGuardrail(guardrail_name="test") + + data = { + "model": "qwen3-reranker", + "query": "What is machine learning?", + "instruction": "Rank by relevance to ML research", + "documents": ["Doc 1", "Doc 2"], + } + + result = await handler.process_input_messages(data, guardrail) + + # Both user-controlled text fields are scanned and written back + assert result["query"] == "What is machine learning? [GUARDRAILED]" + assert result["instruction"] == "Rank by relevance to ML research [GUARDRAILED]" + # Documents unchanged + assert result["documents"] == ["Doc 1", "Doc 2"] + + @pytest.mark.asyncio + async def test_instruction_masked_with_pii(self): + """A masking guardrail rewrites instruction, not just query""" + + class PIIMaskingGuardrail(CustomGuardrail): + async def apply_guardrail( + self, inputs: dict, request_data: dict, input_type: str, **kwargs + ) -> dict: + texts = inputs.get("texts", []) + return {"texts": [t.replace("John Doe", "[NAME_REDACTED]") for t in texts]} + + handler = CohereRerankHandler() + guardrail = PIIMaskingGuardrail(guardrail_name="mask_pii") + + data = { + "model": "qwen3-reranker", + "query": "find records", + "instruction": "prioritize anything authored by John Doe", + "documents": ["Doc 1"], + } + + result = await handler.process_input_messages(data, guardrail) + + # The sensitive value in instruction is sanitized before forwarding + assert "John Doe" not in result["instruction"] + assert "[NAME_REDACTED]" in result["instruction"] + assert result["documents"] == ["Doc 1"] + + @pytest.mark.asyncio + async def test_non_string_instruction_not_scanned(self): + """A non-string instruction is left as-is (only strings are scanned)""" + handler = CohereRerankHandler() + guardrail = MockGuardrail(guardrail_name="test") + + data = { + "model": "qwen3-reranker", + "query": "hello", + "instruction": 12345, # invalid type; backend will reject it + "documents": ["Doc 1"], + } + + result = await handler.process_input_messages(data, guardrail) + + # Query still guardrailed; non-string instruction untouched + assert result["query"] == "hello [GUARDRAILED]" + assert result["instruction"] == 12345 + @pytest.mark.asyncio async def test_process_no_query(self): """Test processing when query is missing""" diff --git a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py index 9e6fa608c50..6425e815db0 100644 --- a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py @@ -4,6 +4,7 @@ import sys import pytest from litellm.llms.hosted_vllm.rerank.transformation import HostedVLLMRerankConfig +from litellm.rerank_api.rerank_utils import get_optional_rerank_params from litellm.types.rerank import ( OptionalRerankParams, RerankBilledUnits, @@ -37,6 +38,54 @@ class TestHostedVLLMRerankTransform: assert params["rank_fields"] == ["field1"] assert params["return_documents"] is True + def test_map_cohere_rerank_params_omits_instruction_when_absent(self): + # Backward-compat: when no instruction is supplied, it must not appear + # in the mapped params (and therefore not in the outgoing request body). + params = self.config.map_cohere_rerank_params( + non_default_params=None, + model=self.model, + drop_params=False, + query="test query", + documents=["doc1", "doc2"], + ) + assert "instruction" not in params + + def test_map_cohere_rerank_params_passes_instruction_when_set(self): + params = self.config.map_cohere_rerank_params( + non_default_params=None, + model=self.model, + drop_params=False, + query="test query", + documents=["doc1", "doc2"], + instruction="Rank by relevance to genomics", + ) + assert params["instruction"] == "Rank by relevance to genomics" + + def test_transform_request_includes_instruction_when_set(self): + body = self.config.transform_rerank_request( + model=self.model, + optional_rerank_params={ + "query": "test query", + "documents": ["doc1", "doc2"], + "instruction": "Rank by relevance to genomics", + }, + headers={}, + ) + assert body["instruction"] == "Rank by relevance to genomics" + + def test_transform_request_omits_instruction_when_absent(self): + # exclude_none must drop the field entirely so the body matches the + # pre-existing (instruction-less) shape exactly. + body = self.config.transform_rerank_request( + model=self.model, + optional_rerank_params={ + "query": "test query", + "documents": ["doc1", "doc2"], + }, + headers={}, + ) + assert "instruction" not in body + def test_map_cohere_rerank_params_raises_on_max_chunks_per_doc(self): with pytest.raises( ValueError, match="Hosted VLLM does not support max_chunks_per_doc" @@ -74,6 +123,7 @@ class TestHostedVLLMRerankTransform: } result = self.config._transform_response(response_dict) assert result.id == "abc123" + assert result.results is not None assert len(result.results) == 2 assert result.results[0]["index"] == 0 assert result.results[0]["relevance_score"] == 0.9 @@ -94,3 +144,32 @@ class TestHostedVLLMRerankTransform: } with pytest.raises(ValueError, match="Missing required fields in the result="): self.config._transform_response(response_dict) + + +class TestGetOptionalRerankParamsInstruction: + """`instruction` is threaded through get_optional_rerank_params only when set.""" + + def setup_method(self): + self.config = HostedVLLMRerankConfig() + self.model = "hosted-vllm-model" + + def test_instruction_threaded_when_set(self): + params = get_optional_rerank_params( + rerank_provider_config=self.config, + model=self.model, + drop_params=False, + query="test query", + documents=["doc1", "doc2"], + instruction="Rank by relevance to genomics", + ) + assert params["instruction"] == "Rank by relevance to genomics" + + def test_instruction_absent_when_not_set(self): + params = get_optional_rerank_params( + rerank_provider_config=self.config, + model=self.model, + drop_params=False, + query="test query", + documents=["doc1", "doc2"], + ) + assert "instruction" not in params From b302c4204a005efdd705633e0a762fea693d436a Mon Sep 17 00:00:00 2001 From: "David J. M. Karlsen" Date: Wed, 24 Jun 2026 13:19:38 +0200 Subject: [PATCH 27/30] fix(github_copilot): synthesize empty choices at the provider seam (#30929) Newer Copilot Claude models (opus-4.7, opus-4.8) return responses with choices=[], either carrying Anthropic-native content blocks or, for the max_tokens=1 probe Claude Code sends, no content at all. github_copilot is dispatched through the OpenAI SDK handler, which calls convert_to_model_response_object directly and never invokes GithubCopilotConfig.transform_response, so the empty-choices guard there surfaced as a 500 Instead of synthesizing choices inside the shared convert_to_model_response_object (which would silently turn empty choices into a fabricated success for every provider), add a no-op transform_parsed_response_dict hook on BaseConfig. GithubCopilotConfig overrides it to synthesize choices from Anthropic-native content, reusing its existing parsing, and the OpenAI SDK handler routes its parsed response through the hook before generic conversion. The core utility keeps treating empty choices as an error for all other providers Fixes: https://github.com/BerriAI/litellm/issues/30927 Signed-off-by: David J. M. Karlsen --- litellm/llms/base_llm/chat/transformation.py | 11 ++ .../github_copilot/chat/transformation.py | 153 ++++++++++-------- litellm/llms/openai/openai.py | 10 +- .../test_convert_dict_to_chat_completion.py | 20 ++- .../test_github_copilot_transformation.py | 104 ++++++++++++ 5 files changed, 224 insertions(+), 74 deletions(-) diff --git a/litellm/llms/base_llm/chat/transformation.py b/litellm/llms/base_llm/chat/transformation.py index 8f9d5cad7c4..4f7e98af780 100644 --- a/litellm/llms/base_llm/chat/transformation.py +++ b/litellm/llms/base_llm/chat/transformation.py @@ -377,6 +377,17 @@ class BaseConfig(ABC): ) -> "ModelResponse": pass + def transform_parsed_response_dict(self, parsed_response: dict) -> dict: + """ + Repair a parsed OpenAI-format response dict before generic conversion. + + Providers routed through the OpenAI SDK handler bypass transform_response, + which calls convert_to_model_response_object directly on the SDK's parsed + output. Override this to normalize a malformed response (e.g. github_copilot + returning empty choices for Anthropic-native Claude responses). + """ + return parsed_response + @abstractmethod def get_error_class( self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] diff --git a/litellm/llms/github_copilot/chat/transformation.py b/litellm/llms/github_copilot/chat/transformation.py index 72dacb59f8a..9880ab1eb6e 100644 --- a/litellm/llms/github_copilot/chat/transformation.py +++ b/litellm/llms/github_copilot/chat/transformation.py @@ -194,6 +194,88 @@ class GithubCopilotConfig(OpenAIConfig): ) return text_content, tool_calls, thinking_blocks + @staticmethod + def _normalize_anthropic_usage(usage: dict) -> dict: + normalized = dict(usage) + if "input_tokens" in usage and "prompt_tokens" not in usage: + normalized["prompt_tokens"] = usage["input_tokens"] + if "output_tokens" in usage and "completion_tokens" not in usage: + normalized["completion_tokens"] = usage["output_tokens"] + if "total_tokens" not in normalized: + normalized["total_tokens"] = normalized.get( + "prompt_tokens", 0 + ) + normalized.get("completion_tokens", 0) + return normalized + + @classmethod + def _synthesize_choices_for_anthropic_native(cls, response_json: dict) -> dict: + """ + Synthesize a `choices` array from an Anthropic-native Copilot response. + + Newer Copilot Claude models (e.g. opus-4.7, opus-4.8) return content + blocks and `stop_reason` without an OpenAI-style `choices` array, and the + max_tokens=1 probe returns no content at all. Returns the response + unchanged when it already carries choices. + + See: https://github.com/BerriAI/litellm/issues/29391 + """ + if response_json.get("choices"): + return response_json + + content = "" + tool_calls: List[ChatCompletionToolCallChunk] = [] + thinking_blocks: Optional[List[Any]] = None + raw_content = response_json.get("content") + if isinstance(raw_content, list): + content, tool_calls, thinking_blocks = cls._parse_anthropic_native_content( + raw_content + ) + elif isinstance(raw_content, str): + content = raw_content + + stop_reason = response_json.get("stop_reason") + finish_reason_map = { + "end_turn": "stop", + "max_tokens": "length", + "stop_sequence": "stop", + "tool_use": "tool_calls", + } + if tool_calls: + finish_reason = "tool_calls" + elif stop_reason in finish_reason_map: + finish_reason = finish_reason_map[stop_reason] + elif content: + finish_reason = "stop" + else: + finish_reason = "length" + + message: dict = { + "role": "assistant", + "content": content if content or not tool_calls else None, + } + if tool_calls: + message["tool_calls"] = tool_calls + if thinking_blocks: + message["thinking_blocks"] = thinking_blocks + + synthesized = { + **response_json, + "choices": [ + {"index": 0, "message": message, "finish_reason": finish_reason} + ], + } + usage = response_json.get("usage") + if isinstance(usage, dict): + synthesized["usage"] = cls._normalize_anthropic_usage(usage) + return synthesized + + def transform_parsed_response_dict(self, parsed_response: dict) -> dict: + """ + Repair the OpenAI-SDK-parsed response on the handler path that bypasses + transform_response. See: https://github.com/BerriAI/litellm/issues/30927 + """ + return self._synthesize_choices_for_anthropic_native(parsed_response) + def transform_response( self, model: str, @@ -208,15 +290,6 @@ class GithubCopilotConfig(OpenAIConfig): api_key: Optional[str] = None, json_mode: Optional[bool] = None, ) -> "ModelResponse": - """ - Handle newer Copilot models (e.g. claude-opus-4.7, claude-opus-4.8) that - return Anthropic-native format responses without a `choices` array. - - Synthesizes the missing `choices` from Anthropic-native fields, then - delegates to the parent so all standard post-processing applies. - - See: https://github.com/BerriAI/litellm/issues/29391 - """ try: response_json = raw_response.json() except Exception: @@ -235,70 +308,12 @@ class GithubCopilotConfig(OpenAIConfig): ) if not response_json.get("choices"): - content = "" - tool_calls: List[ChatCompletionToolCallChunk] = [] - thinking_blocks: Optional[List[Any]] = None - if "content" in response_json and isinstance( - response_json["content"], list - ): - content, tool_calls, thinking_blocks = ( - self._parse_anthropic_native_content(response_json["content"]) - ) - elif isinstance(response_json.get("content"), str): - content = response_json["content"] - - stop_reason = response_json.get("stop_reason") - finish_reason_map = { - "end_turn": "stop", - "max_tokens": "length", - "stop_sequence": "stop", - "tool_use": "tool_calls", - } - # Prefer tool_calls when blocks were extracted; otherwise map stop_reason. - if tool_calls: - finish_reason = "tool_calls" - elif stop_reason in finish_reason_map: - finish_reason = finish_reason_map[stop_reason] - elif content: - finish_reason = "stop" - else: - finish_reason = "length" - - message: dict = { - "role": "assistant", - "content": content if content or not tool_calls else None, - } - if tool_calls: - message["tool_calls"] = tool_calls - if thinking_blocks: - message["thinking_blocks"] = thinking_blocks - - response_json["choices"] = [ - { - "index": 0, - "message": message, - "finish_reason": finish_reason, - } - ] - - if "usage" in response_json: - usage = response_json["usage"] - if "input_tokens" in usage and "prompt_tokens" not in usage: - usage["prompt_tokens"] = usage["input_tokens"] - if "output_tokens" in usage and "completion_tokens" not in usage: - usage["completion_tokens"] = usage["output_tokens"] - if "total_tokens" not in usage: - usage["total_tokens"] = usage.get("prompt_tokens", 0) + usage.get( - "completion_tokens", 0 - ) - - # Build a patched response so super() sees valid JSON with choices - patched = httpx.Response( + response_json = self._synthesize_choices_for_anthropic_native(response_json) + raw_response = httpx.Response( status_code=raw_response.status_code, headers=raw_response.headers, content=json.dumps(response_json).encode(), ) - raw_response = patched return super().transform_response( model=model, diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index ea905d8ebca..8237aaa010a 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -785,7 +785,11 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): ) logging_obj.model_call_details["response_headers"] = headers - stringified_response = response.model_dump() + stringified_response = ( + provider_config.transform_parsed_response_dict( + response.model_dump() + ) + ) logging_obj.post_call( input=messages, api_key=api_key, @@ -933,7 +937,9 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): timeout=timeout, logging_obj=logging_obj, ) - stringified_response = response.model_dump() + stringified_response = provider_config.transform_parsed_response_dict( + response.model_dump() + ) logging_obj.post_call( input=data["messages"], api_key=api_key, diff --git a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py index 9a69f513069..d46436f209b 100644 --- a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py +++ b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py @@ -1627,7 +1627,12 @@ class TestMissingChoicesGuard: assert "no 'choices'" in exc_info.value.message def test_convert_to_model_response_object_empty_choices_raises_api_error(self): - """Empty choices list raises APIError.""" + """Empty choices list raises APIError, same as missing/null choices. + + Provider-specific repair (e.g. github_copilot synthesizing choices for + Anthropic-native responses) happens before this guard, in the provider + config; the core utility keeps treating empty choices as an error. + """ from litellm.exceptions import APIError response_object = { @@ -1683,7 +1688,9 @@ class TestMissingChoicesGuard: assert "no 'choices'" in exc_info.value.message - def test_convert_to_model_response_object_stream_true_no_choices_raises_api_error(self): + def test_convert_to_model_response_object_stream_true_no_choices_raises_api_error( + self, + ): """Missing choices via stream=True path raises APIError when generator is consumed.""" from litellm.exceptions import APIError @@ -2471,6 +2478,13 @@ class TestConvertToModelResponseObjectCompletion: def test_model_response_none_raises(self): with pytest.raises(Exception): convert_to_model_response_object( - response_object={"choices": [{"message": {"content": "hi", "role": "assistant"}, "finish_reason": "stop"}]}, + response_object={ + "choices": [ + { + "message": {"content": "hi", "role": "assistant"}, + "finish_reason": "stop", + } + ] + }, model_response_object=None, ) diff --git a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py index 5673ad81551..f69ba7df938 100644 --- a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py +++ b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py @@ -878,3 +878,107 @@ class TestGithubCopilotTransformResponse: litellm_params={}, encoding=None, ) + + +class TestGithubCopilotTransformParsedResponseDict: + """ + Tests for GithubCopilotConfig.transform_parsed_response_dict, the hook the + OpenAI SDK handler calls on its parsed response. That handler bypasses + transform_response, so this is the seam that repairs empty-choices responses + from newer Copilot Claude models on the live completion path. + + See: https://github.com/BerriAI/litellm/issues/30927 + """ + + def test_synthesizes_choices_from_anthropic_content(self): + config = GithubCopilotConfig() + + parsed = { + "id": "msg_vrtx_01", + "model": "claude-opus-4.8", + "object": "chat.completion", + "choices": [], + "content": [{"type": "text", "text": "Hello!"}], + "stop_reason": "end_turn", + "usage": {"input_tokens": 10, "output_tokens": 5}, + } + + repaired = config.transform_parsed_response_dict(parsed) + + assert len(repaired["choices"]) == 1 + choice = repaired["choices"][0] + assert choice["message"]["content"] == "Hello!" + assert choice["finish_reason"] == "stop" + assert repaired["usage"]["prompt_tokens"] == 10 + assert repaired["usage"]["completion_tokens"] == 5 + assert repaired["usage"]["total_tokens"] == 15 + + def test_passthrough_when_choices_present(self): + config = GithubCopilotConfig() + + parsed = { + "id": "chatcmpl-1", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok"}, + "finish_reason": "stop", + } + ], + } + + assert config.transform_parsed_response_dict(parsed) is parsed + + +@patch("litellm.llms.openai.openai.OpenAIChatCompletion._get_openai_client") +@patch( + "litellm.llms.openai.openai.OpenAIChatCompletion.make_sync_openai_chat_completion_request" +) +def test_openai_handler_repairs_github_copilot_empty_choices( + mock_request, mock_get_client +): + """ + The OpenAI SDK handler calls convert_to_model_response_object directly on the + SDK's parsed output, bypassing transform_response. convert raises APIError on + empty choices, so the handler must route github_copilot responses through + transform_parsed_response_dict first. Removing that wiring (or resolving a + config without the override) fails this test with APIError. + + See: https://github.com/BerriAI/litellm/issues/30927 + """ + from litellm.llms.openai.openai import OpenAIChatCompletion + + mock_get_client.return_value = MagicMock() + + class _FakeSDKResponse: + def model_dump(self): + return { + "id": "msg_vrtx_01", + "model": "claude-opus-4.8", + "object": "chat.completion", + "choices": [], + "content": [{"type": "text", "text": "Hi there"}], + "stop_reason": "end_turn", + "usage": {"input_tokens": 12, "output_tokens": 3}, + } + + mock_request.return_value = ({}, _FakeSDKResponse()) + + result = OpenAIChatCompletion().completion( + model="claude-opus-4.8", + messages=[{"role": "user", "content": "Hi"}], + model_response=ModelResponse(), + timeout=60.0, + optional_params={}, + litellm_params={}, + logging_obj=MagicMock(), + custom_llm_provider="github_copilot", + client=MagicMock(), + api_key="gh.test-key-123456789", + acompletion=False, + ) + + assert isinstance(result, ModelResponse) + assert result.choices[0].message.content == "Hi there" + assert result.choices[0].finish_reason == "stop" + mock_request.assert_called_once() From 2f6fd18842bde8bc3cccb15373b0fe1fdc6b358b Mon Sep 17 00:00:00 2001 From: Vedant Agarwal <43557509+Vedant-Agarwal@users.noreply.github.com> Date: Wed, 24 Jun 2026 19:20:59 +0800 Subject: [PATCH 28/30] fix(router): stop fallback lookups from mutating the router fallbacks config (#30624) * fix: correct amazon.titan-embed-text-v2 input price to $0.02/1M tokens (#29693) * fix: correct amazon.titan-embed-text-v2 input price to $0.02/1M tokens * test: scope local cost map env var with monkeypatch to avoid test pollution * fix(sensitive_data_masker): fully mask secrets at or below the reveal threshold (#30764) * fix(sensitive_data_masker): fully mask secrets at or below the reveal threshold _mask_value did partial reveal by showing the first visible_prefix and last visible_suffix characters, but for a value whose length was at or below visible_prefix + visible_suffix (8 by default) it returned the value verbatim. A value of exactly 8 chars fell through the length guard and computed masked_length == 0, reconstructing the original string with no mask characters; anything shorter hit the early return. Either way short credentials were emitted in plaintext. mask_dict routes real secrets through this path, so an 8-char-or-shorter redis password, api key, or token could be written to logs and the UI unmasked. The sibling helper mask_sensitive_keys already guards this case; _mask_value now does the same by fully masking any value at or below the threshold. * fix(sensitive_data_masker): add mask_short_values opt-out for truncation callers Fully masking short values is the right default for secret masking, but CooldownCache reuses the masker purely to truncate exception messages to the first 50 characters, and it relies on short messages being returned readable. Masking those blanked out short exception text and broke its tests. Add a mask_short_values flag (default True, secure) and have CooldownCache pass False so it keeps the truncation behavior, while every secret-masking caller still gets short values fully masked. * fix(mcp_debug): opt out of short-value masking to keep diagnostic token preview MCPDebug uses the masker to preview auth tokens in debug headers and documents that values of 10 chars or fewer are shown unchanged so token types stay distinguishable. Pass mask_short_values=False so that diagnostic behavior is preserved while secret maskers keep masking short values. * fix(mcp_debug): mask short auth values in debug headers instead of echoing them Earlier this masker opted out of short-value masking to keep a token preview, but that echoes short authorization and token values verbatim in debug response headers, which is the same leak this change is meant to close. Auth material should never be emitted in full, so mask short values here too; the first/last character preview still applies to longer tokens. Only CooldownCache keeps the opt-out, since it truncates exception text rather than masking secrets. * test(mcp_debug): assert masked short value preserves length * refactor(fireworks_ai): remove deprecated audio transcriptions endpoint (#30917) Fireworks AI deprecated audio inference on 2026-06-10 (https://docs.fireworks.ai/updates/changelog#audio-inference-and-image-generation-deprecation). Live API testing confirms the endpoint is already non-functional: a valid Fireworks API key receives HTTP 401 "Unauthorized" from api.fireworks.ai/inference/v1/audio/transcriptions for every request, regardless of payload. The audio-prod.api.fireworks.ai host referenced in the test suite returns 401 for every path; the entire host is decommissioned. Remove the dead FireworksAIAudioTranscriptionConfig class and every reference to it across the codebase: - Delete litellm/llms/fireworks_ai/audio_transcription/ directory (17-line config class that inherited from OpenAIWhisperAudioTranscriptionConfig) - Remove the Fireworks branch from ProviderConfigManager.get_provider_audio_transcription_config() in litellm/utils.py; update the stale comment in get_optional_params_transcription that referenced fireworks ai - Remove the FireworksAIAudioTranscriptionConfig entries from LLM_CONFIG_NAMES and _LLM_CONFIGS_IMPORT_MAP in litellm/_lazy_imports_registry.py - Remove the TYPE_CHECKING re-export in litellm/__init__.py - Remove the transcription branch in the fireworks_ai case of get_supported_openai_params() in litellm/litellm_core_utils/get_supported_openai_params.py - Remove the whisper-v3 and whisper-v3-turbo entries from model_prices_and_context_window.json and litellm/model_prices_and_context_window_backup.json (both had mode: audio_transcription and zero-cost pricing) - Remove the TestFireworksAIAudioTranscription test class and its imports from tests/llm_translation/test_fireworks_ai_translation.py No other provider is affected. The openai_compatible_providers list, FireworksAIMixin, and the OpenAI Whisper transcription handler all stay because they are shared with other Fireworks endpoints and other providers. The provider_endpoints_support.json registry already had audio_transcriptions set to false for fireworks_ai. * feat: add darkbloom provider (#30876) * feat: add darkbloom provider * fix: document darkbloom provider endpoints * fix: address darkbloom review feedback * fix: update darkbloom tool metadata * fix: fail fast for non-Postgres database URLs (#30883) * fix(proxy): fail fast on non-PostgreSQL DATABASE_URL instead of hanging on startup LiteLLM's Prisma datasource is pinned to provider = 'postgresql', so a sqlite:// or mysql:// DATABASE_URL can never connect. Today that surfaces as an opaque startup stall where the port never binds, and a separate 'DB not connected' 500 on /key/generate when no DATABASE_URL is set at all leaves operators guessing what to configure. Validate the DATABASE_URL / DIRECT_URL scheme in run_server before any Prisma call and exit with an actionable message naming the unsupported scheme. Also reword CommonProxyErrors.db_not_connected_error to tell the operator to set DATABASE_URL to a postgresql:// connection string. Add regression tests covering postgres acceptance and sqlite/mysql/mssql rejection. * fix: resolve CI failures and proxy DB URL typing issue * fix(proxy): fail fast on non-PostgreSQL DATABASE_URLs with clear startup errors instead of hanging * Validate DIRECT_URL alongside DATABASE_URL startup guards * fix(bedrock): surface modeled HTTP status for mid-stream error events so 5xx is retryable (#24608) (#30946) * fix(bedrock): surface modeled HTTP status for mid-stream error events (#24608) * test(bedrock): mid-stream server errors trigger streaming fallback (#24608) * style(bedrock): black-format stream-error helper (#24608) * fix(mcp): re-land native tool preservation with typed annotations (#30645) * fix(mcp): preserve native tools in semantic filter hook with typed annotations * fix(mcp): tighten _is_mcp_tool Chat Completions shape check * fix(sambanova): return embeddings supported params instead of dropping them (#30937) * fix(router): send fallback metadata when streaming (#30914) When a streaming request triggers a fallback, there was previously no way to know it happened. This commit addresses this in a few ways: 1. The response now correctly populates the fallback headers (`x-litellm-attempted-fallbacks`) so callers know a fallback happened. 2. The correct model ID is passed in the streaming chunks. 3. A streaming chunk with the fallback error can be optionally sent back to the client (opt-in) by passing `include_fallback_errors: true` in the request. The format of the fallback errors while streaming is intentionally OpenAI compatible to not break existing libraries that parse these events. It was tested with Vercel's AI SDK (ai-sdk.dev). It is also opt-in, so it is not delieved unexpectedly to callers by default. * fix(mistral): drop output-only reasoning fields from input messages (#30884) LiteLLM attaches reasoning_content and thinking_blocks to assistant responses. Replaying those assistant turns verbatim forwarded the fields back to Mistral, whose input schema forbids unknown keys, so the whole request failed with a 422 extra_forbidden and reasoning models became unusable across multiple turns. Strip both fields from assistant messages before the request is built, in a spot that runs ahead of the image/file branch so it applies on every path. Fixes #30835 Co-authored-by: Cursor * fix(perplexity): bill search queries at the per-request price, not 1/1000 of it (#30652) * fix(perplexity): bill search queries at the per-request price, not 1/1000 The fallback cost calculator divided search_context_cost_per_query by 1000, but that field stores the per-request price in USD: sonar is {low: 0.005, medium: 0.008, high: 0.012}, matching Perplexity's published $5/$8/$12 per 1,000 requests expressed per request. The gemini cost calculator reads the same field per request with no division (its docstring calls it "the per-request cost"). The division understated search cost by 1000x on every Perplexity call that falls back to manual calculation (i.e. when the API does not return a pre-computed usage.cost). Use the value directly. Update the tests that had encoded the /1000 factor in their expectations, and drop an unused import flagged by ruff in the touched test file. * test(perplexity): update integration test search-cost expectations to per-request The integration tests still encoded the old /1000 search-cost factor, so they failed once the fallback calculator was corrected to bill search_context_cost_per_query per request. Update the four expected-cost computations (and the high-volume dollar-value comments) to match. * test(perplexity): drop unused mock imports flagged by ruff * fix: include model_access_groups when expanding all-team-models in get_team_models (#30622) * fix(fireworks_ai): return None for transcription in get_supported_openai_params Fireworks AI deprecated audio inference on 2026-06-10; the endpoint is decommissioned. Without an explicit transcription branch, requests with request_type='transcription' fell through to the else and returned FireworksAIConfig chat-completion params. Return None instead to signal the provider does not support transcription. * fix(proxy): gate include_fallback_errors behind expose_fallback_errors_to_caller setting Without an operator gate, any authenticated caller could set include_fallback_errors=True, trigger a fallback, and read raw upstream exception messages from the x-litellm-fallback-errors header and the litellm-fallback-metadata SSE event. Strip include_fallback_errors from request data in common_processing_pre_call_logic when expose_fallback_errors_to_caller is not set, so the router never builds the error list. Also gate _should_include_fallback_errors on the same setting as a secondary check for the streaming SSE injection path. * test(proxy): opt in to expose_fallback_errors_to_caller in streaming SSE test The operator gate added in e7ff3e1 means include_fallback_errors is only honoured when general_settings.expose_fallback_errors_to_caller is True. Set that flag via monkeypatch in the test that exercises the emit path. * test(prompt_templates): make test_convert_url hermetic instead of hitting picsum.photos test_convert_url called convert_url_to_base64 against a live picsum.photos URL and asserted nothing, so it added no real signal and broke CI whenever the host was unreachable (it was returning 522 and blocking this branch). Replace the live call with a mocked HTTP client and assert the produced base64 data URL, so the conversion path is exercised deterministically with no network dependency. This suite runs under VCR, which is why a transport level mock (respx) does not reliably intercept; mocking the client object itself is robust regardless. * fix(interactions): drop role from Interaction response to match Google spec Google removed the output-only role field from the Interaction schema (it now lives only on Turn), so the live OpenAPI compliance canary started failing with 'role' not in spec. Reconcile our generated types by removing role from Interaction, CreateModelInteractionParams, CreateAgentInteractionParams and from the LiteLLM InteractionsAPIResponse/InteractionsAPIStreamingResponse, stop stamping role=model in the responses-to-interactions transformation, and update the compliance and integration tests accordingly. Turn.role is kept since the spec still defines it. * fix: align all-team-models sentinel access * fix(router): forward include_fallback_errors through multi-hop fallbacks run_async_fallback received include_fallback_errors as an explicit named parameter, so it was bound out of **kwargs and never reached the nested async_function_with_fallbacks call. Multi-hop fallback chains (a fallback group that itself fails over) therefore stopped collecting fallback errors beyond the first hop when a caller opted in. Re-inject the flag into kwargs before the nested call so inner hops keep accumulating errors, which add_fallback_headers_to_response already merges across levels. * fix(router): stop fallback lookups from mutating the router fallbacks config get_fallback_model_group resolved a bare-string fallback by popping it out of the fallbacks list it was handed. That list is frequently the live router.fallbacks config, so a single lookup permanently removed the entry and the configured fallback stopped applying to later requests until restart. The pop also ran inside enumerate(), shifting indices and skipping an adjacent string fallback. Read the item instead of popping it, and add a regression test that fails on the old mutating behavior --------- Co-authored-by: Srivatsa Kamballa Co-authored-by: Ahmad Shahzad <107808273+shzdehmd@users.noreply.github.com> Co-authored-by: Jeremy Chapeau <113923302+jychp@users.noreply.github.com> Co-authored-by: KRISH SONI <67964054+krishvsoni@users.noreply.github.com> Co-authored-by: Kent <72616338+kingdoooo@users.noreply.github.com> Co-authored-by: Ayush Shekhar <106994833+ayushh0110@users.noreply.github.com> Co-authored-by: dav nguyxn Co-authored-by: Tal Marian Co-authored-by: Hemant K <51333870+hemant1026@users.noreply.github.com> Co-authored-by: Cursor Co-authored-by: Yash Raj Pandey <55940078+devYRPauli@users.noreply.github.com> Co-authored-by: Zang Peiyu <166481866+factnn@users.noreply.github.com> Co-authored-by: Sameer Kankute Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- .../router_utils/fallback_event_handlers.py | 2 +- .../test_fallback_event_handlers.py | 18 +++++++++++++++++- 2 files changed, 18 insertions(+), 2 deletions(-) diff --git a/litellm/router_utils/fallback_event_handlers.py b/litellm/router_utils/fallback_event_handlers.py index f0edc7fc9db..891d80d785a 100644 --- a/litellm/router_utils/fallback_event_handlers.py +++ b/litellm/router_utils/fallback_event_handlers.py @@ -72,7 +72,7 @@ def get_fallback_model_group( elif list(item.keys())[0] == "*": # check generic fallback generic_fallback_idx = idx elif isinstance(item, str): - fallback_model_group = [fallbacks.pop(idx)] # returns single-item list + fallback_model_group = [item] ## if none, check for generic fallback if fallback_model_group is None: if stripped_model_fallback is not None: diff --git a/tests/test_litellm/router_utils/test_fallback_event_handlers.py b/tests/test_litellm/router_utils/test_fallback_event_handlers.py index ca647bdce55..98a34de295c 100644 --- a/tests/test_litellm/router_utils/test_fallback_event_handlers.py +++ b/tests/test_litellm/router_utils/test_fallback_event_handlers.py @@ -2,7 +2,10 @@ import json import pytest -from litellm.router_utils.fallback_event_handlers import run_async_fallback +from litellm.router_utils.fallback_event_handlers import ( + get_fallback_model_group, + run_async_fallback, +) class StreamingWrapper: @@ -137,3 +140,16 @@ async def test_run_async_fallback_skips_original_model_group(): ) assert response._hidden_params["additional_headers"]["x-litellm-attempted-fallbacks"] == 1 + + +def test_get_fallback_model_group_does_not_mutate_fallbacks(): + """A string fallback must be resolved without mutating the caller's + fallbacks list, which is the live router config shared across requests.""" + fallbacks = [{"gpt-3.5-turbo": ["claude-3-haiku"]}, "gpt-4o-mini"] + + fallback_model_group, _ = get_fallback_model_group( + fallbacks=fallbacks, model_group="unmatched-model" + ) + + assert fallback_model_group == ["gpt-4o-mini"] + assert fallbacks == [{"gpt-3.5-turbo": ["claude-3-haiku"]}, "gpt-4o-mini"] From af7b0af52fa1383ceb5804b6c92a8e724b52947c Mon Sep 17 00:00:00 2001 From: bhumikadangayach <139267865+bhumikadangayach@users.noreply.github.com> Date: Wed, 24 Jun 2026 16:57:41 +0530 Subject: [PATCH 29/30] fix(sambanova): update pricing, deprecate retired models, and add missing models (#30016) * feat(bedrock): add amazon.titan-embed-g1-text-02 embedding model support - Add model to provider routing allowlist in embedding.py - Add request transformation using AmazonTitanG1Config - Add response transformation using AmazonTitanG1Config - Add pricing metadata to model_prices_and_context_window.json - Add unit tests for embedding and model info Fixes missing cost tracking reported in #29786 Related to VANDRANKI/litellm PR #29790 * style: fix syntax error, trailing whitespace and missing newline * style: apply black formatting to embedding.py * style: apply black formatting to test_bedrock_embedding.py * fix(sambanova): update pricing, fix context windows, add deprecation dates, and add missing models * fix(sambanova): sync model_prices_and_context_window_backup.json with primary * fix(sambanova): fix indentation on Meta-Llama-3.2-1B-Instruct deprecation_date * fix(bedrock): add amazon.titan-embed-g1-text-02 to unmapped model error message * style: apply black formatting to embedding.py * fix(sambanova): correct indentation on DeepSeek-V3.2 entry * fix(sambanova): replace gemma-3-12b-it with gemma-4-31B-it (verified pricing) --- litellm/llms/bedrock/embed/embedding.py | 10 + ...odel_prices_and_context_window_backup.json | 987 +++++++++++++----- model_prices_and_context_window.json | 57 +- .../llm_translation/test_bedrock_embedding.py | 15 + 4 files changed, 819 insertions(+), 250 deletions(-) diff --git a/litellm/llms/bedrock/embed/embedding.py b/litellm/llms/bedrock/embed/embedding.py index b6aa99842d7..98c2d87bfdb 100644 --- a/litellm/llms/bedrock/embed/embedding.py +++ b/litellm/llms/bedrock/embed/embedding.py @@ -224,6 +224,10 @@ class BedrockEmbedding(BaseAWSLLM): returned_response = AmazonTitanV2Config()._transform_response( response_list=response_list, model=model ) + elif model == "amazon.titan-embed-g1-text-02": + returned_response = AmazonTitanG1Config()._transform_response( + response_list=response_list, model=model + ) elif provider == "twelvelabs": returned_response = ( TwelveLabsMarengoEmbeddingConfig()._transform_response( @@ -447,6 +451,7 @@ class BedrockEmbedding(BaseAWSLLM): "amazon.titan-embed-image-v1", "amazon.titan-embed-text-v1", "amazon.titan-embed-text-v2:0", + "amazon.titan-embed-g1-text-02", ]: batch_data = [] for i in input: @@ -464,6 +469,10 @@ class BedrockEmbedding(BaseAWSLLM): transformed_request = AmazonTitanV2Config()._transform_request( input=i, inference_params=inference_params ) + elif model == "amazon.titan-embed-g1-text-02": + transformed_request = AmazonTitanG1Config()._transform_request( + input=i, inference_params=inference_params + ) else: raise Exception( "Unmapped model. Received={}. Expected={}".format( @@ -472,6 +481,7 @@ class BedrockEmbedding(BaseAWSLLM): "amazon.titan-embed-image-v1", "amazon.titan-embed-text-v1", "amazon.titan-embed-text-v2:0", + "amazon.titan-embed-g1-text-02", ], ) ) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6ebac7efc8d..452a33be695 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -570,6 +570,15 @@ "output_cost_per_token": 0.0, "output_vector_size": 1536 }, + "amazon.titan-embed-g1-text-02": { + "input_cost_per_token": 1e-07, + "litellm_provider": "bedrock", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "embedding", + "output_cost_per_token": 0.0, + "output_vector_size": 1536 + }, "amazon.titan-embed-text-v2:0": { "input_cost_per_token": 2e-08, "litellm_provider": "bedrock", @@ -14143,6 +14152,14 @@ "notes": "APISerpent deep search (/api/search), multi-engine (Google, Bing, Yahoo, DuckDuckGo). Pricing: $0.60/1k searches." } }, + "tinyfish/search": { + "input_cost_per_query": 0.0, + "litellm_provider": "tinyfish", + "mode": "search", + "metadata": { + "notes": "TinyFish Search API" + } + }, "elevenlabs/scribe_v1": { "input_cost_per_second": 6.11e-05, "litellm_provider": "elevenlabs", @@ -31357,13 +31374,13 @@ "output_cost_per_token": 0.0 }, "sambanova/MiniMax-M2.7": { - "input_cost_per_token": 3e-07, + "input_cost_per_token": 6e-07, "litellm_provider": "sambanova", - "max_input_tokens": 204800, + "max_input_tokens": 196608, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 1.2e-06, + "output_cost_per_token": 2.4e-06, "source": "https://cloud.sambanova.ai/plans/pricing", "supports_function_calling": true, "supports_reasoning": true, @@ -31380,6 +31397,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/DeepSeek-R1-Distill-Llama-70B": { + "deprecation_date": "2026-03-20", "input_cost_per_token": 7e-07, "litellm_provider": "sambanova", "max_input_tokens": 131072, @@ -31390,6 +31408,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/DeepSeek-V3-0324": { + "deprecation_date": "2026-04-14", "input_cost_per_token": 3e-06, "litellm_provider": "sambanova", "max_input_tokens": 32768, @@ -31420,6 +31439,7 @@ "supports_vision": true }, "sambanova/Llama-4-Scout-17B-16E-Instruct": { + "deprecation_date": "2025-06-19", "input_cost_per_token": 4e-07, "litellm_provider": "sambanova", "max_input_tokens": 8192, @@ -31436,6 +31456,7 @@ "supports_tool_choice": true }, "sambanova/Meta-Llama-3.1-405B-Instruct": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 5e-06, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31449,6 +31470,7 @@ "supports_tool_choice": true }, "sambanova/Meta-Llama-3.1-8B-Instruct": { + "deprecation_date": "2026-04-14", "input_cost_per_token": 1e-07, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31462,6 +31484,7 @@ "supports_tool_choice": true }, "sambanova/Meta-Llama-3.2-1B-Instruct": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 4e-08, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31472,6 +31495,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/Meta-Llama-3.2-3B-Instruct": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 8e-08, "litellm_provider": "sambanova", "max_input_tokens": 4096, @@ -31495,6 +31519,7 @@ "supports_tool_choice": true }, "sambanova/Meta-Llama-Guard-3-8B": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 3e-07, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31505,6 +31530,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/QwQ-32B": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 5e-07, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31515,6 +31541,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/Qwen2-Audio-7B-Instruct": { + "deprecation_date": "2025-06-19", "input_cost_per_token": 5e-07, "litellm_provider": "sambanova", "max_input_tokens": 4096, @@ -31526,6 +31553,7 @@ "supports_audio_input": true }, "sambanova/Qwen3-32B": { + "deprecation_date": "2026-04-06", "input_cost_per_token": 4e-07, "litellm_provider": "sambanova", "max_input_tokens": 8192, @@ -31539,9 +31567,9 @@ "supports_tool_choice": true }, "sambanova/DeepSeek-V3.1": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, "input_cost_per_token": 3e-06, "output_cost_per_token": 4.5e-06, "litellm_provider": "sambanova", @@ -31555,8 +31583,8 @@ "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 3e-06, - "output_cost_per_token": 4.5e-06, + "input_cost_per_token": 2.2e-07, + "output_cost_per_token": 5.9e-07, "litellm_provider": "sambanova", "mode": "chat", "supports_function_calling": true, @@ -31564,21 +31592,55 @@ "supports_reasoning": true, "source": "https://cloud.sambanova.ai/plans/pricing" }, - "snowflake/claude-3-5-sonnet": { - "litellm_provider": "snowflake", - "max_input_tokens": 18000, - "max_output_tokens": 8192, - "max_tokens": 8192, - "mode": "chat", - "supports_computer_use": true - }, - "snowflake/deepseek-r1": { - "litellm_provider": "snowflake", + "sambanova/DeepSeek-V3.2": { + "max_tokens": 32768, "max_input_tokens": 32768, - "max_output_tokens": 8192, - "max_tokens": 8192, + "max_output_tokens": 32768, + "input_cost_per_token": 3e-06, + "output_cost_per_token": 4.5e-06, + "litellm_provider": "sambanova", "mode": "chat", - "supports_reasoning": true + "supports_function_calling": true, + "supports_tool_choice": true, + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/gemma-4-31B-it": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.8e-07, + "output_cost_per_token": 1.15e-06, + "litellm_provider": "sambanova", + "mode": "chat", + "supports_vision": true, + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "snowflake/claude-3-5-sonnet": { + "litellm_provider": "snowflake", + "max_input_tokens": 200000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "input_cost_per_token": 0.000003, + "output_cost_per_token": 0.000015, + "cache_read_input_token_cost": 0.0000003, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_response_schema": true + }, + "snowflake/deepseek-r1": { + "litellm_provider": "snowflake", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "input_cost_per_token": 0.00000135, + "output_cost_per_token": 0.0000054, + "supports_reasoning": true, + "supports_system_messages": true }, "snowflake/gemma-7b": { "litellm_provider": "snowflake", @@ -31632,23 +31694,34 @@ "snowflake/llama3.1-405b": { "litellm_provider": "snowflake", "max_input_tokens": 128000, - "max_output_tokens": 8192, - "max_tokens": 8192, - "mode": "chat" + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "input_cost_per_token": 0.0000012, + "output_cost_per_token": 0.0000012, + "supports_function_calling": true, + "supports_system_messages": true }, "snowflake/llama3.1-70b": { "litellm_provider": "snowflake", "max_input_tokens": 128000, - "max_output_tokens": 8192, - "max_tokens": 8192, - "mode": "chat" + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "input_cost_per_token": 0.00000072, + "output_cost_per_token": 0.00000072, + "supports_function_calling": true, + "supports_system_messages": true }, "snowflake/llama3.1-8b": { "litellm_provider": "snowflake", "max_input_tokens": 128000, - "max_output_tokens": 8192, - "max_tokens": 8192, - "mode": "chat" + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "input_cost_per_token": 0.00000024, + "output_cost_per_token": 0.00000024, + "supports_system_messages": true }, "snowflake/llama3.2-1b": { "litellm_provider": "snowflake", @@ -31664,13 +31737,17 @@ "max_tokens": 8192, "mode": "chat" }, - "snowflake/llama3.3-70b": { - "litellm_provider": "snowflake", + "snowflake/llama3.3-70b": { + "max_tokens": 16384, "max_input_tokens": 128000, - "max_output_tokens": 8192, - "max_tokens": 8192, - "mode": "chat" - }, + "max_output_tokens": 16384, + "input_cost_per_token": 0.00000072, + "output_cost_per_token": 0.00000072, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_system_messages": true + }, "snowflake/mistral-7b": { "litellm_provider": "snowflake", "max_input_tokens": 32000, @@ -31685,12 +31762,17 @@ "max_tokens": 8192, "mode": "chat" }, - "snowflake/mistral-large2": { + "snowflake/mistral-large2": { "litellm_provider": "snowflake", "max_input_tokens": 128000, - "max_output_tokens": 8192, - "max_tokens": 8192, - "mode": "chat" + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.000006, + "supports_function_calling": true, + "supports_system_messages": true, + "supports_response_schema": true }, "snowflake/mixtral-8x7b": { "litellm_provider": "snowflake", @@ -31727,13 +31809,17 @@ "max_tokens": 8192, "mode": "chat" }, - "snowflake/snowflake-llama-3.3-70b": { + "snowflake/snowflake-llama-3.3-70b": { + "max_tokens": 16384, + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.00000072, + "output_cost_per_token": 0.00000072, "litellm_provider": "snowflake", - "max_input_tokens": 8000, - "max_output_tokens": 8192, - "max_tokens": 8192, - "mode": "chat" - }, + "mode": "chat", + "supports_function_calling": true, + "supports_system_messages": true + }, "stability/sd3": { "litellm_provider": "stability", "mode": "image_generation", @@ -32076,6 +32162,11 @@ "litellm_provider": "tavily", "mode": "search" }, + "you_com/search": { + "input_cost_per_query": 0.0, + "litellm_provider": "you_com", + "mode": "search" + }, "text-completion-codestral/codestral-2405": { "input_cost_per_token": 0.0, "litellm_provider": "text-completion-codestral", @@ -36595,17 +36686,7 @@ "max_input_tokens": 32000, "max_tokens": 32000, "mode": "embedding", - "output_cost_per_token": 0.0, - "supports_vision": true - }, - "voyage/voyage-multimodal-3.5": { - "input_cost_per_token": 1.2e-07, - "litellm_provider": "voyage", - "max_input_tokens": 32000, - "max_tokens": 32000, - "mode": "embedding", - "output_cost_per_token": 0.0, - "supports_vision": true + "output_cost_per_token": 0.0 }, "wandb/openai/gpt-oss-120b": { "max_tokens": 131072, @@ -40270,6 +40351,178 @@ "supports_tool_choice": true, "supports_vision": true }, + "scaleway/qwen/qwen3.5-397b-a17b": { + "input_cost_per_token": 6e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 256000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 3.6e-06, + "supports_function_calling": true, + "supports_reasoning": true, + "supports_vision": true + }, + "scaleway/qwen/qwen3.6-35b-a3b": { + "input_cost_per_token": 2.5e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 256000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "supports_function_calling": true, + "supports_vision": true, + "supports_reasoning": true + }, + "scaleway/qwen/qwen3-235b-a22b-instruct-2507": { + "input_cost_per_token": 7.5e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 256000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 2.25e-06, + "supports_function_calling": true + }, + "scaleway/qwen/qwen3-embedding-8b": { + "input_cost_per_token": 1e-07, + "litellm_provider": "scaleway", + "mode": "embedding", + "output_cost_per_token": 0.0 + }, + "scaleway/qwen/qwen3-coder-30b-a3b-instruct": { + "input_cost_per_token": 2e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 128000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 8e-07, + "supports_function_calling": true + }, + "scaleway/openai/gpt-oss-120b": { + "input_cost_per_token": 1.5e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 128000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6e-07, + "supports_function_calling": true + }, + "scaleway/openai/whisper-large-v3": { + "input_cost_per_audio_token": 0.0, + "litellm_provider": "scaleway", + "mode": "audio_transcription", + "output_cost_per_token": 0.0 + }, + "scaleway/google/gemma-4-26b-a4b-it": { + "input_cost_per_token": 2.5e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 256000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 5e-07, + "supports_function_calling": true, + "supports_reasoning": true, + "supports_vision": true + }, + "scaleway/google/gemma-3-27b-it": { + "input_cost_per_token": 2.5e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 40000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 5e-07, + "supports_function_calling": true, + "supports_vision": true + }, + "scaleway/hcompany/holo2-30b-a3b": { + "input_cost_per_token": 3e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 22000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 7e-07, + "supports_reasoning": true, + "supports_vision": true + }, + "scaleway/mistralai/mistral-medium-3.5-128b": { + "input_cost_per_token": 1.5e-06, + "litellm_provider": "scaleway", + "max_input_tokens": 256000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 7.5e-06, + "supports_reasoning": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_tool_choice": true + }, + "scaleway/mistralai/devstral-2-123b-instruct-2512": { + "input_cost_per_token": 4e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 200000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 2e-06, + "supports_function_calling": true + }, + "scaleway/mistralai/voxtral-small-24b-2507": { + "input_cost_per_audio_token": 1.5e-07, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 32000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 3.5e-07, + "supports_audio_input": true + }, + "scaleway/mistralai/mistral-small-3.2-24b-instruct-2506": { + "input_cost_per_token": 1.5e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 128000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3.5e-07, + "supports_function_calling": true, + "supports_vision": true + }, + "scaleway/mistralai/pixtral-12b-2409": { + "input_cost_per_token": 2e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 128000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "chat", + "output_cost_per_token": 2e-07, + "supports_vision": true, + "supports_function_calling": true + }, + "scaleway/BAAI/bge-multilingual-gemma2": { + "input_cost_per_token": 1e-07, + "litellm_provider": "scaleway", + "mode": "embedding", + "output_cost_per_token": 0.0 + }, + "scaleway/meta/llama-3.3-70b-instruct": { + "input_cost_per_token": 9e-07, + "litellm_provider": "scaleway", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 9e-07, + "supports_function_calling": true + }, "novita/deepseek/deepseek-v3.2": { "litellm_provider": "novita", "mode": "chat", @@ -43051,192 +43304,362 @@ "supports_native_structured_output": true, "supports_pdf_input": true }, - "soniox/stt-async-v4": { - "litellm_provider": "soniox", - "max_output_tokens": 8000, - "max_tokens": 8000, - "input_cost_per_second": 0.0, - "output_cost_per_second": 0.0000277778, - "mode": "audio_transcription", - "source": "https://soniox.com/pricing", - "supported_endpoints": [ - "/v1/audio/transcriptions" - ], - "supports_audio_input": true - }, - "soniox/stt-async-v5": { - "litellm_provider": "soniox", - "max_output_tokens": 8000, - "max_tokens": 8000, - "input_cost_per_second": 0.0, - "output_cost_per_second": 0.0000277778, - "mode": "audio_transcription", - "source": "https://soniox.com/pricing", - "supported_endpoints": [ - "/v1/audio/transcriptions" - ], - "supports_audio_input": true - }, - "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 6e-07, - "output_cost_per_token": 3.6e-06, - "cache_read_input_token_cost": 0, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 4.5e-07, - "output_cost_per_token": 1.8e-06, - "cache_read_input_token_cost": 0, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/Qwen/Qwen3.6-27B-FP8": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 3.2e-07, - "output_cost_per_token": 3.2e-06, - "cache_read_input_token_cost": 0, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/lukealonso/GLM-5.1-NVFP4-MTP": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 1.4e-06, - "output_cost_per_token": 4.4e-06, - "cache_read_input_token_cost": 0, - "max_input_tokens": 202752, - "max_output_tokens": 202752, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/deepseek-ai/DeepSeek-V4-Flash": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 1.4e-07, - "output_cost_per_token": 2.8e-07, - "cache_read_input_token_cost": 0, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/moonshotai/Kimi-K2.6": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 9.6e-07, - "output_cost_per_token": 4e-06, - "cache_read_input_token_cost": 0, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/MiniMaxAI/MiniMax-M2.5": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 3e-07, - "output_cost_per_token": 1.2e-06, - "cache_read_input_token_cost": 0, - "max_input_tokens": 196608, - "max_output_tokens": 196608, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/google/gemma-4-31B-it": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 1.4e-07, - "output_cost_per_token": 5.6e-07, - "cache_read_input_token_cost": 0, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/openai/gpt-oss-120b": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 1.5e-07, - "output_cost_per_token": 6e-07, - "cache_read_input_token_cost": 0, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - }, - "tensormesh/openai/gpt-oss-20b": { - "litellm_provider": "tensormesh", - "mode": "chat", - "input_cost_per_token": 7e-08, - "output_cost_per_token": 2.8e-07, - "cache_read_input_token_cost": 0, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "supports_function_calling": true, - "supports_tool_choice": true, - "supports_response_schema": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_reasoning": true, - "source": "https://serverless.tensormesh.ai/v1/models/openrouter" - } -, + "snowflake/claude-sonnet-4-5": { + "max_tokens": 16384, + "max_input_tokens": 200000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.000003, + "output_cost_per_token": 0.000015, + "cache_read_input_token_cost": 0.0000003, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_response_schema": true + }, + "snowflake/claude-sonnet-4-6": { + "max_tokens": 16384, + "max_input_tokens": 200000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.000003, + "output_cost_per_token": 0.000015, + "cache_read_input_token_cost": 0.0000003, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_response_schema": true + }, + "snowflake/claude-4-sonnet": { + "max_tokens": 16384, + "max_input_tokens": 200000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.000003, + "output_cost_per_token": 0.000015, + "cache_read_input_token_cost": 0.0000003, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_response_schema": true + }, + "snowflake/claude-4-opus": { + "max_tokens": 16384, + "max_input_tokens": 200000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.000005, + "output_cost_per_token": 0.000025, + "cache_read_input_token_cost": 0.0000005, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_response_schema": true + }, + "snowflake/claude-haiku-4-5": { + "max_tokens": 16384, + "max_input_tokens": 200000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.000001, + "output_cost_per_token": 0.000005, + "cache_read_input_token_cost": 0.0000001, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_response_schema": true + }, + "snowflake/claude-3-7-sonnet": { + "max_tokens": 16384, + "max_input_tokens": 200000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.000003, + "output_cost_per_token": 0.000015, + "cache_read_input_token_cost": 0.0000003, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_response_schema": true + }, + "snowflake/openai-gpt-4.1": { + "max_tokens": 16384, + "max_input_tokens": 300000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.000008, + "cache_read_input_token_cost": 0.0000005, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_response_schema": true + }, + "snowflake/openai-gpt-5": { + "max_tokens": 16384, + "max_input_tokens": 300000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.00000125, + "output_cost_per_token": 0.00001, + "cache_read_input_token_cost": 0.000000125, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "supports_response_schema": true + }, + "snowflake/openai-gpt-5-mini": { + "max_tokens": 16384, + "max_input_tokens": 1000000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.0000003, + "output_cost_per_token": 0.0000012, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_system_messages": true, + "supports_response_schema": true + }, + "snowflake/openai-gpt-5-nano": { + "max_tokens": 16384, + "max_input_tokens": 5000000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.00000015, + "output_cost_per_token": 0.0000006, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_system_messages": true, + "supports_response_schema": true + }, + "snowflake/llama4-maverick": { + "max_tokens": 16384, + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.00000024, + "output_cost_per_token": 0.00000097, + "litellm_provider": "snowflake", + "mode": "chat", + "supports_function_calling": true, + "supports_system_messages": true + }, + "snowflake/snowflake-arctic-embed-l-v2.0": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "input_cost_per_token": 0.00000007, + "output_cost_per_token": 0.0, + "litellm_provider": "snowflake", + "mode": "embedding" + }, + "snowflake/snowflake-arctic-embed-m-v2.0": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "input_cost_per_token": 0.00000007, + "output_cost_per_token": 0.0, + "litellm_provider": "snowflake", + "mode": "embedding" + }, + "soniox/stt-async-v4": { + "litellm_provider": "soniox", + "max_output_tokens": 8000, + "max_tokens": 8000, + "input_cost_per_second": 0.0, + "output_cost_per_second": 0.0000277778, + "mode": "audio_transcription", + "source": "https://soniox.com/pricing", + "supported_endpoints": ["/v1/audio/transcriptions"], + "supports_audio_input": true + }, + "soniox/stt-async-v5": { + "litellm_provider": "soniox", + "max_output_tokens": 8000, + "max_tokens": 8000, + "input_cost_per_second": 0.0, + "output_cost_per_second": 0.0000277778, + "mode": "audio_transcription", + "source": "https://soniox.com/pricing", + "supported_endpoints": ["/v1/audio/transcriptions"], + "supports_audio_input": true + }, + "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 6e-07, + "output_cost_per_token": 3.6e-06, + "cache_read_input_token_cost": 0, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 4.5e-07, + "output_cost_per_token": 1.8e-06, + "cache_read_input_token_cost": 0, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/Qwen/Qwen3.6-27B-FP8": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 3.2e-07, + "output_cost_per_token": 3.2e-06, + "cache_read_input_token_cost": 0, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/lukealonso/GLM-5.1-NVFP4-MTP": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 1.4e-06, + "output_cost_per_token": 4.4e-06, + "cache_read_input_token_cost": 0, + "max_input_tokens": 202752, + "max_output_tokens": 202752, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/deepseek-ai/DeepSeek-V4-Flash": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 1.4e-07, + "output_cost_per_token": 2.8e-07, + "cache_read_input_token_cost": 0, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/moonshotai/Kimi-K2.6": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 9.6e-07, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 0, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/MiniMaxAI/MiniMax-M2.5": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1.2e-06, + "cache_read_input_token_cost": 0, + "max_input_tokens": 196608, + "max_output_tokens": 196608, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/google/gemma-4-31B-it": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 1.4e-07, + "output_cost_per_token": 5.6e-07, + "cache_read_input_token_cost": 0, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/openai/gpt-oss-120b": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 0, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + }, + "tensormesh/openai/gpt-oss-20b": { + "litellm_provider": "tensormesh", + "mode": "chat", + "input_cost_per_token": 7e-08, + "output_cost_per_token": 2.8e-07, + "cache_read_input_token_cost": 0, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_reasoning": true, + "source": "https://serverless.tensormesh.ai/v1/models/openrouter" + } + , "deepseek-v4-flash": { "cache_creation_input_token_cost": 0.0, "cache_read_input_token_cost": 2.8e-09, @@ -43370,5 +43793,83 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": false + }, + "pinstripes/ps/glm-4.5-air": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 0.000000125, + "output_cost_per_token": 0.00000045, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": true, + "source": "https://pinstripes.io/pricing" + }, + "pinstripes/ps/qwen3.6-35b-a3b": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.00000014, + "output_cost_per_token": 0.00000045, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": true, + "source": "https://pinstripes.io/pricing" + }, + "pinstripes/ps/qwen3-30b-a3b": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.00000009, + "output_cost_per_token": 0.0000002, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": true, + "source": "https://pinstripes.io/pricing" + }, + "pinstripes/ps/qwen3-coder-30b-a3b": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.0000003, + "output_cost_per_token": 0.0000006, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": false, + "source": "https://pinstripes.io/pricing" + }, + "pinstripes/ps/deepseek-v4-flash": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 0.0000001, + "output_cost_per_token": 0.0000002, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": true, + "source": "https://pinstripes.io/pricing" + }, + "pinstripes/ps/minimax-m2.7": { + "max_tokens": 1000192, + "max_input_tokens": 1000192, + "max_output_tokens": 1000192, + "input_cost_per_token": 0.000000255, + "output_cost_per_token": 0.00000055, + "litellm_provider": "pinstripes", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_reasoning": false, + "source": "https://pinstripes.io/pricing" } } diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d7017a40993..cffb00c6b22 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -570,6 +570,15 @@ "output_cost_per_token": 0.0, "output_vector_size": 1536 }, + "amazon.titan-embed-g1-text-02": { + "input_cost_per_token": 1e-07, + "litellm_provider": "bedrock", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "embedding", + "output_cost_per_token": 0.0, + "output_vector_size": 1536 + }, "amazon.titan-embed-text-v2:0": { "input_cost_per_token": 2e-08, "litellm_provider": "bedrock", @@ -31381,13 +31390,13 @@ "output_cost_per_token": 0.0 }, "sambanova/MiniMax-M2.7": { - "input_cost_per_token": 3e-07, + "input_cost_per_token": 6e-07, "litellm_provider": "sambanova", - "max_input_tokens": 204800, + "max_input_tokens": 196608, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 1.2e-06, + "output_cost_per_token": 2.4e-06, "source": "https://cloud.sambanova.ai/plans/pricing", "supports_function_calling": true, "supports_reasoning": true, @@ -31404,6 +31413,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/DeepSeek-R1-Distill-Llama-70B": { + "deprecation_date": "2026-03-20", "input_cost_per_token": 7e-07, "litellm_provider": "sambanova", "max_input_tokens": 131072, @@ -31414,6 +31424,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/DeepSeek-V3-0324": { + "deprecation_date": "2026-04-14", "input_cost_per_token": 3e-06, "litellm_provider": "sambanova", "max_input_tokens": 32768, @@ -31444,6 +31455,7 @@ "supports_vision": true }, "sambanova/Llama-4-Scout-17B-16E-Instruct": { + "deprecation_date": "2025-06-19", "input_cost_per_token": 4e-07, "litellm_provider": "sambanova", "max_input_tokens": 8192, @@ -31460,6 +31472,7 @@ "supports_tool_choice": true }, "sambanova/Meta-Llama-3.1-405B-Instruct": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 5e-06, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31473,6 +31486,7 @@ "supports_tool_choice": true }, "sambanova/Meta-Llama-3.1-8B-Instruct": { + "deprecation_date": "2026-04-14", "input_cost_per_token": 1e-07, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31486,6 +31500,7 @@ "supports_tool_choice": true }, "sambanova/Meta-Llama-3.2-1B-Instruct": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 4e-08, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31496,6 +31511,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/Meta-Llama-3.2-3B-Instruct": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 8e-08, "litellm_provider": "sambanova", "max_input_tokens": 4096, @@ -31519,6 +31535,7 @@ "supports_tool_choice": true }, "sambanova/Meta-Llama-Guard-3-8B": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 3e-07, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31529,6 +31546,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/QwQ-32B": { + "deprecation_date": "2025-06-25", "input_cost_per_token": 5e-07, "litellm_provider": "sambanova", "max_input_tokens": 16384, @@ -31539,6 +31557,7 @@ "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/Qwen2-Audio-7B-Instruct": { + "deprecation_date": "2025-06-19", "input_cost_per_token": 5e-07, "litellm_provider": "sambanova", "max_input_tokens": 4096, @@ -31550,6 +31569,7 @@ "supports_audio_input": true }, "sambanova/Qwen3-32B": { + "deprecation_date": "2026-04-06", "input_cost_per_token": 4e-07, "litellm_provider": "sambanova", "max_input_tokens": 8192, @@ -31563,9 +31583,9 @@ "supports_tool_choice": true }, "sambanova/DeepSeek-V3.1": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, "input_cost_per_token": 3e-06, "output_cost_per_token": 4.5e-06, "litellm_provider": "sambanova", @@ -31579,13 +31599,36 @@ "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, + "input_cost_per_token": 2.2e-07, + "output_cost_per_token": 5.9e-07, + "litellm_provider": "sambanova", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/DeepSeek-V3.2": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, "input_cost_per_token": 3e-06, "output_cost_per_token": 4.5e-06, "litellm_provider": "sambanova", "mode": "chat", "supports_function_calling": true, "supports_tool_choice": true, - "supports_reasoning": true, + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/gemma-4-31B-it": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.8e-07, + "output_cost_per_token": 1.15e-06, + "litellm_provider": "sambanova", + "mode": "chat", + "supports_vision": true, "source": "https://cloud.sambanova.ai/plans/pricing" }, "snowflake/claude-3-5-sonnet": { diff --git a/tests/llm_translation/test_bedrock_embedding.py b/tests/llm_translation/test_bedrock_embedding.py index 92c22f582d9..2bc4192833b 100644 --- a/tests/llm_translation/test_bedrock_embedding.py +++ b/tests/llm_translation/test_bedrock_embedding.py @@ -34,6 +34,11 @@ img_base_64 = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAGQAAABkBAMAAACCzIh "text", titan_embedding_response, ), # V2 text model + ( + "bedrock/amazon.titan-embed-g1-text-02", + "text", + titan_embedding_response, + ), # G1 text model ( "bedrock/amazon.titan-embed-image-v1", "image", @@ -459,3 +464,13 @@ def test_bedrock_embedding_region_bug_reproduction(): os.environ["AWS_REGION_NAME"] = original_region_name else: os.environ.pop("AWS_REGION_NAME", None) + + +def test_bedrock_titan_g1_text_02_model_info(): + """Test that amazon.titan-embed-g1-text-02 has correct pricing metadata""" + model_info = litellm.get_model_info("amazon.titan-embed-g1-text-02") + assert model_info is not None, "Model info should not be None" + assert model_info["litellm_provider"] == "bedrock" + assert model_info["mode"] == "embedding" + assert model_info["input_cost_per_token"] == 1e-07 + assert model_info["max_input_tokens"] == 8192 From 6bf3b9b6db64d80bd91c1a59c8b2bb334d7d2f87 Mon Sep 17 00:00:00 2001 From: Ewertonslv Date: Wed, 24 Jun 2026 08:29:25 -0300 Subject: [PATCH 30/30] fix(utils): preserve arbitrary above-threshold tiered pricing keys in get_model_info (#30880) * fix(utils): preserve arbitrary above-threshold tiered pricing keys in get_model_info get_model_info rebuilt ModelInfo by copying a fixed allow-list of input/output_cost_per_token_above__tokens keys (128k/200k/272k/512k), so any other threshold a user registered was dropped before reaching _get_token_base_cost, which already reads an arbitrary threshold out of the key name. Custom tiers such as above_500k_tokens were silently ignored and billing fell back to the base per-token rate. Carry over any _above__tokens cost key present on the source cost-map entry that the fixed fields miss Fixes #30344 * test(cost): keep suite hermetic by popping the temp tiered-pricing model Wrap the regression body in try/finally so litellm.model_cost no longer leaks the litellm-test-non-standard-tier entry into later tests that iterate or reset the global cost map. Addresses Greptile review thread. --- litellm/utils.py | 12 +++++- .../llm_cost_calc/test_llm_cost_calc_utils.py | 38 +++++++++++++++++++ 2 files changed, 49 insertions(+), 1 deletion(-) diff --git a/litellm/utils.py b/litellm/utils.py index 5c3ab3e1490..0a7dd1a1b8f 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -5844,6 +5844,9 @@ def _is_potential_model_name_in_model_cost( ) +_ABOVE_THRESHOLD_COST_KEY = re.compile(r"_above_\d+k?_tokens$") + + def _get_model_info_helper( model: str, custom_llm_provider: Optional[str] = None, @@ -6021,7 +6024,7 @@ def _get_model_info_helper( ) _output_cost_per_token = 0 - return ModelInfoBase( + returned_model_info = ModelInfoBase( key=key, max_tokens=_model_info.get("max_tokens", None), max_input_tokens=_model_info.get("max_input_tokens", None), @@ -6238,6 +6241,13 @@ def _get_model_info_helper( uses_embed_content=_model_info.get("uses_embed_content", None), supports_image_size=_model_info.get("supports_image_size", None), ) + for cost_key, cost_value in _model_info.items(): + if ( + cost_key not in returned_model_info + and _ABOVE_THRESHOLD_COST_KEY.search(cost_key) is not None + ): + returned_model_info[cost_key] = cost_value # type: ignore[literal-required] + return returned_model_info except Exception as e: verbose_logger.debug(f"Error getting model info: {e}") raise Exception( diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index 7f3d5a959a1..d47558d302d 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -384,6 +384,44 @@ def test_generic_cost_per_token_minimax_m3_above_512k_tokens(): assert round(completion_cost, 10) == round(expected_completion, 10) +def test_generic_cost_per_token_honors_non_standard_above_threshold(): + """Regression for #30344: get_model_info must keep arbitrary + input/output_cost_per_token_above__tokens thresholds, not only the hard-coded + 128k/200k/272k/512k set, so a custom tier boundary is applied past its limit.""" + model = "litellm-test-non-standard-tier" + custom_llm_provider = "openai" + litellm.register_model( + { + model: { + "litellm_provider": custom_llm_provider, + "mode": "chat", + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "input_cost_per_token_above_500k_tokens": 9e-6, + "output_cost_per_token_above_500k_tokens": 18e-6, + } + } + ) + + try: + prompt_tokens = 600000 + completion_tokens = 1000 + usage = Usage( + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + total_tokens=prompt_tokens + completion_tokens, + ) + prompt_cost, completion_cost = generic_cost_per_token( + model=model, + usage=usage, + custom_llm_provider=custom_llm_provider, + ) + assert round(prompt_cost, 10) == round(9e-6 * prompt_tokens, 10) + assert round(completion_cost, 10) == round(18e-6 * completion_tokens, 10) + finally: + litellm.model_cost.pop(model, None) + + def test_generic_cost_per_token_gpt55(): """gpt-5.5: base pricing — $5/1M input, $30/1M output, $0.50/1M cached input.""" model = "gpt-5.5"