chore: merge main into litellm_vertex_context_cache_creation_accounting

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Devin AI 2026-09-23 03:37:57 +00:00
commit 3e5a79b5f4
95 changed files with 5085 additions and 1045 deletions

View file

@ -0,0 +1,33 @@
name: Compat Matrix Image
on:
pull_request:
paths:
- tests/e2e/claude_code/cron_vm/**
- .github/workflows/compat-matrix-image.yml
workflow_dispatch:
permissions: {}
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
compat-matrix-image:
name: compat-matrix-image
runs-on: ubuntu-latest
timeout-minutes: 15
permissions:
contents: read
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Build the Render cron image
run: docker build -f tests/e2e/claude_code/cron_vm/Dockerfile -t compat-matrix:${{ github.sha }} tests/e2e
- name: Run the pinned binaries as the cron user
run: |
docker run --rm compat-matrix:${{ github.sha }} bash -c 'set -e; whoami; claude --version; gh --version; uv --version'

View file

@ -6,8 +6,12 @@ on:
workflow_dispatch:
inputs:
issue_number:
description: "Closed issue number to comment on manually."
required: true
description: "Closed issue number to comment on and close the superseded pull requests of. Ignored by a sweep."
required: false
sweep:
description: "Close every open pull request whose linked issues were all fixed on the default branch. Reads every open pull request, so run it at most once an hour."
type: boolean
default: false
pull_request:
paths:
- .github/workflows/issue_fixed_comment.yml
@ -39,16 +43,17 @@ jobs:
with:
bun-version: "1.4.0"
- name: Test the closer lookup, the release placement and the comment
- name: Test the closer lookup, the release placement, the comment and the superseded pull request close
run: bun test scripts/comment-fixed-issue.test.ts
comment-fixed-issue:
if: github.event_name != 'pull_request' && github.repository == 'BerriAI/litellm'
runs-on: ubuntu-latest
timeout-minutes: 5
timeout-minutes: 15
permissions:
contents: read
issues: write
pull-requests: write
steps:
- name: Checkout scripts
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
@ -59,13 +64,16 @@ jobs:
- name: Setup Bun
uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with:
# Exact version, never latest: the next step holds an issues: write token
# Exact version, never latest: the next step holds issues: write and pull-requests: write tokens
bun-version: "1.4.0"
- name: Name the release that carries the fix
- name: Name the release that carries the fix and close the pull requests it supersedes
shell: bash
run: bun run scripts/comment-fixed-issue.ts | tee -a "${GITHUB_STEP_SUMMARY}"
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
ISSUE_NUMBER: ${{ github.event.issue.number || github.event.inputs.issue_number }}
SWEEP: ${{ github.event.inputs.sweep }}
DEFAULT_BRANCH: ${{ github.event.repository.default_branch }}
DRY_RUN: ${{ vars.ISSUE_FIXED_COMMENT_ENABLED != 'true' }}
CLOSE_PRS_DRY_RUN: ${{ vars.ISSUE_FIXED_CLOSE_PRS_ENABLED != 'true' }}

View file

@ -105,6 +105,16 @@ jobs:
with:
python-version: "3.12"
- uses: ./.github/actions/setup-uv-with-retries
with:
version: "0.10.9"
- name: Install Python dependencies for the bridge tests
working-directory: .
run: |
uv sync --frozen --no-install-project
echo "PYTHONPATH=$PWD/.venv/lib/$(ls .venv/lib)/site-packages" >> "$GITHUB_ENV"
- run: rustup toolchain install --no-self-update
- uses: taiki-e/install-action@d438492cf8a250514fa2d34b30bc3c0dc37c65ff # v2.87.8

View file

@ -1,10 +1,10 @@
# syntax=docker/dockerfile:1.7
# Base image for building
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
# Runtime image
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43

View file

@ -1,5 +1,5 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin

View file

@ -1,10 +1,10 @@
# syntax=docker/dockerfile:1.7
# Base image for building
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
# Runtime image
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43

View file

@ -1,8 +1,8 @@
# syntax=docker/dockerfile:1.7
# Base images
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG PROXY_EXTRAS_SOURCE=published
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.

View file

@ -1,6 +1,6 @@
[project]
name = "litellm-enterprise"
version = "0.1.69"
version = "0.1.70"
description = "Package for LiteLLM Enterprise features"
readme = "README.md"
requires-python = ">=3.9"
@ -26,7 +26,7 @@ required-version = ">=0.10.9"
module-root = ""
[tool.commitizen]
version = "0.1.69"
version = "0.1.70"
version_files = [
"pyproject.toml:^version",
"../pyproject.toml:litellm-enterprise==",

View file

@ -1,5 +1,5 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x)
ARG PGBOUNCER_VERSION=1.25.2

View file

@ -1,6 +1,6 @@
[project]
name = "litellm-proxy-extras"
version = "0.4.100"
version = "0.4.101"
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
readme = "README.md"
requires-python = ">=3.9"
@ -26,7 +26,7 @@ required-version = ">=0.10.9"
module-root = ""
[tool.commitizen]
version = "0.4.100"
version = "0.4.101"
version_files = [
"pyproject.toml:^version",
"../pyproject.toml:litellm-proxy-extras==",

View file

@ -2946,6 +2946,7 @@ name = "litellm-core-utils"
version = "0.1.0"
dependencies = [
"fancy-regex 0.19.2",
"litellm-tracing",
"litellm-types",
"rstest",
"serde",
@ -3075,6 +3076,7 @@ dependencies = [
"aws-sdk-secretsmanager",
"bytes",
"criterion",
"fancy-regex 0.19.2",
"futures-util",
"litellm-auth",
"litellm-auth-aws",
@ -3093,6 +3095,7 @@ dependencies = [
"litellm-callbacks-legacy-python",
"litellm-core",
"litellm-core-utils",
"litellm-host",
"litellm-host-python",
"litellm-http",
"litellm-llms",
@ -3100,6 +3103,7 @@ dependencies = [
"litellm-secrets-aws",
"litellm-secrets-types",
"litellm-token-counter",
"litellm-tracing",
"litellm-types",
"pyo3",
"pyo3-async-runtimes",
@ -3172,11 +3176,11 @@ dependencies = [
"litellm-auth-aws",
"litellm-core-utils",
"litellm-secrets-types",
"litellm-tracing",
"rstest",
"serde_json",
"thiserror 2.0.19",
"tokio",
"tracing",
"veil",
"wiremock",
]
@ -3208,6 +3212,7 @@ dependencies = [
"base64 0.22.1",
"litellm-core-utils",
"litellm-secrets-types",
"litellm-tracing",
"moka",
"percent-encoding",
"reqwest 0.12.28",
@ -3216,7 +3221,6 @@ dependencies = [
"serde_json",
"thiserror 2.0.19",
"tokio",
"tracing",
"veil",
"wiremock",
]
@ -3331,6 +3335,19 @@ dependencies = [
"tiktoken-rs",
]
[[package]]
name = "litellm-tracing"
version = "0.1.0"
dependencies = [
"fancy-regex 0.19.2",
"percent-encoding",
"rstest",
"serde_json",
"tokio",
"tracing",
"tracing-subscriber",
]
[[package]]
name = "litellm-types"
version = "0.1.0"

View file

@ -9,6 +9,8 @@ license = "MIT"
repository = "https://github.com/BerriAI/litellm"
[workspace.dependencies]
litellm-tracing = { path = "crates/tracing" }
tracing = "0.1"
litellm-core = { path = "crates/core" }
litellm-host = { path = "crates/host" }
litellm-callbacks-legacy-python = { path = "crates/callbacks-legacy-python" }

View file

@ -7,6 +7,7 @@ repository.workspace = true
[dependencies]
fancy-regex.workspace = true
litellm-tracing.workspace = true
litellm-types.workspace = true
serde.workspace = true
serde_json.workspace = true

View file

@ -1,109 +1 @@
use fancy_regex::Regex;
pub const REDACTED: &str = "REDACTED";
const DEFAULT_MINIMUM_CUSTOM_KEY_LENGTH: usize = 16;
fn minimum_custom_key_length() -> usize {
std::env::var("MINIMUM_CUSTOM_KEY_LENGTH")
.ok()
.and_then(|value| value.trim().parse().ok())
.unwrap_or(DEFAULT_MINIMUM_CUSTOM_KEY_LENGTH)
}
fn secret_patterns(minimum_custom_key_length: usize) -> String {
let sk_suffix_length = minimum_custom_key_length.saturating_sub("sk-".len());
[
r"-----BEGIN[A-Z \-]*PRIVATE KEY-----[\s\S]*?-----END[A-Z \-]*PRIVATE KEY-----",
r"\bya29\.[A-Za-z0-9_.~+/-]+",
r#"(?:client_secret|azure_password|azure_username)\s+[^\s,'"})\]{}>]+"#,
r"(?:AKIA|ASIA)[0-9A-Z]{16}",
r"Bearer\s+[A-Za-z0-9\-._~+/]{10,}=*",
r"Basic\s+[A-Za-z0-9+/]{10,}={0,2}",
&format!(r"sk-[A-Za-z0-9\-_]{{{sk_suffix_length},}}"),
r#"(?<=[?&])(?:api[_-]?key|\w*(?:token|password|passwd|client_secret|secret_key|_secret))=[^\s&'"]+"#,
r#"(?:api[_-]?key)['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]{8,}"#,
r#"(?:x-api-key|api-key)['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]+"#,
r"x-ak-[A-Za-z0-9\-_]{20,}",
r"AIza[0-9A-Za-z\-_]{35}",
r#"(?<=[?&])key=[^\s&'"]{8,}"#,
r#"(?:^|(?<=\W))\w*(?:password|passwd|client_secret|secret_key|_secret)['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]+"#,
r#"(?<=://)[^\s'":]{0,4096}:[^\s'"]{1,4096}(?=@)"#,
r"dapi[0-9a-f]{32}",
r#"litellm\.[A-Za-z0-9_]*_key['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]+"#,
r#"private_key['"]?\s*[:=]\s*['"]?(?:-----BEGIN[A-Z \-]*PRIVATE KEY-----[\s\S]*?-----END[A-Z \-]*PRIVATE KEY-----|[^\s,'"})\]{}>]+)"#,
concat!(
r"(?:master_key|xai_key|database_url|db_url|connection_string|",
r"aws_secret_access_key|aws_session_token|aws_access_key_id|",
r"signing_key|encryption_key|",
r"auth_token|access_token|refresh_token|",
r"slack_webhook_url|webhook_url|",
r"database_connection_string|",
r"huggingface_token|jwt_secret)",
r#"['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]+"#,
),
r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]*",
r"(?<=[?&])sig=[A-Za-z0-9%+/=]+",
r#"\{[^{}]*"type"\s*:\s*"service_account"[^{}]*(?:\{[^{}]*\}[^{}]*)*\}"#,
]
.join("|")
}
/// Python's `_ENABLE_SECRET_REDACTION` pattern set, compiled once per configuration.
#[derive(Clone, Debug)]
pub struct SecretRedactor {
pattern: Regex,
}
impl SecretRedactor {
pub fn new(minimum_custom_key_length: usize) -> Self {
let pattern = Regex::new(&format!(
"(?i){}",
secret_patterns(minimum_custom_key_length)
))
.expect("secret redaction patterns compile");
Self { pattern }
}
/// `None` when `LITELLM_DISABLE_REDACT_SECRETS` turns redaction off.
pub fn from_env() -> Option<Self> {
let disabled = std::env::var("LITELLM_DISABLE_REDACT_SECRETS")
.is_ok_and(|value| value.eq_ignore_ascii_case("true"));
(!disabled).then(|| Self::new(minimum_custom_key_length()))
}
pub fn redact(&self, value: &str) -> String {
self.pattern.replace_all(value, REDACTED).into_owned()
}
}
#[cfg(test)]
mod tests {
use super::*;
#[rstest::rstest]
#[case::bearer("auth failed: Bearer abcdefghijklmnop", "auth failed: REDACTED")]
#[case::sk_key("key sk-abcdefghijklmnopqrstuvwxyz rejected", "key REDACTED rejected")]
#[case::short_sk_key_is_kept("sk-abc", "sk-abc")]
#[case::query_param("GET /v1?api_key=secret123&x=1", "GET /v1?REDACTED&x=1")]
#[case::dict_repr("{'api_key': 'abcdefghij'}", "{'REDACTED'}")]
#[case::url_credentials("postgres://user:pass@host/db", "postgres://REDACTED@host/db")]
#[case::case_insensitive("BEARER ABCDEFGHIJKLMNOP", "REDACTED")]
#[case::aws_key("AKIAABCDEFGHIJKLMNOP", "REDACTED")]
#[case::sas_signature("https://x.blob/a?sv=1&sig=abc%2B=", "https://x.blob/a?sv=1&REDACTED")]
#[case::password_needs_word_boundary("db_password=hunter2", "REDACTED")]
#[case::plain_text_is_kept(r#"{"message": "rejected"}"#, r#"{"message": "rejected"}"#)]
fn redacts_the_same_spans_as_the_python_patterns(#[case] input: &str, #[case] expected: &str) {
assert_eq!(
SecretRedactor::new(DEFAULT_MINIMUM_CUSTOM_KEY_LENGTH).redact(input),
expected
);
}
#[test]
fn sk_threshold_follows_the_minimum_custom_key_length() {
let redactor = SecretRedactor::new(8);
assert_eq!(redactor.redact("sk-abcde"), REDACTED);
assert_eq!(redactor.redact("sk-abcd"), "sk-abcd");
}
}
pub use litellm_tracing::{REDACTED, SecretRedactor};

View file

@ -19,6 +19,9 @@ huggingface = ["litellm-token-counter/huggingface"]
tiktoken = ["litellm-token-counter/tiktoken"]
[dependencies]
fancy-regex.workspace = true
litellm-tracing.workspace = true
litellm-host.workspace = true
bytes.workspace = true
futures-util.workspace = true
litellm-cache.workspace = true

View file

@ -1,6 +1,7 @@
use crate::logger::run_sync_value;
use litellm_cache_gcs::{DEFAULT_ENDPOINT, GcsConfig};
use litellm_cache_redis_semantic::RedisSemanticConfig;
use litellm_host_python::{release_gil, run_sync_value};
use litellm_host_python::release_gil;
use litellm_http::ClientVariant;
use pyo3::prelude::*;

View file

@ -1,5 +1,6 @@
use crate::logger::run_async;
use litellm_cache_response::PartialHits;
use litellm_host_python::{ExecutionStep, from_py, release_gil, run_async, to_py};
use litellm_host_python::{ExecutionStep, from_py, release_gil, to_py};
use pyo3::{
PyTraverseError, PyVisit,
exceptions::{PyRuntimeError, PyValueError},

View file

@ -1,10 +1,11 @@
use crate::logger::run_sync_value;
use litellm_auth_aws::AwsAuthConfig;
use litellm_cache_gcs::{DEFAULT_ENDPOINT, GcsConfig};
use litellm_cache_qdrant_semantic::{OpenAiEmbedderConfig, Quantization};
use litellm_cache_redis::{RedisNode, RedisTopology};
use litellm_cache_redis_semantic::RedisSemanticConfig;
use litellm_cache_s3::{S3CacheConfig, S3Endpoint};
use litellm_host_python::{release_gil, run_sync_value};
use litellm_host_python::release_gil;
use litellm_http::ClientVariant;
use pyo3::{
PyTraverseError, PyVisit,

View file

@ -470,7 +470,7 @@ impl NativeResponseCache {
match self {
Self::Exact(_) | Self::QdrantSemantic(_) => {
let service = self.clone();
litellm_host_python::run_async(
crate::logger::run_async(
py,
async move {
service
@ -495,7 +495,7 @@ impl NativeResponseCache {
match self {
Self::Exact(_) | Self::QdrantSemantic(_) => {
let service = self.clone();
litellm_host_python::run_async(
crate::logger::run_async(
py,
async move { service.async_lookup(&request, now()).await },
super::cache_error,
@ -550,7 +550,7 @@ impl NativeResponseCache {
match self {
Self::Exact(_) | Self::QdrantSemantic(_) => {
let service = self.clone();
litellm_host_python::run_async(
crate::logger::run_async(
py,
async move { service.async_store(&request, response, now()).await },
super::cache_error,
@ -619,7 +619,7 @@ impl NativeResponseCache {
match self {
Self::Exact(_) | Self::QdrantSemantic(_) => {
let service = self.clone();
litellm_host_python::run_async(
crate::logger::run_async(
py,
async move { service.async_store_batch(entries, now()).await },
super::cache_error,

View file

@ -1,7 +1,8 @@
use crate::logger::run_async;
use std::{collections::VecDeque, time::Duration};
use litellm_cache::Error;
use litellm_host_python::{Execution, ExecutionBody, ExecutionStep, run_async};
use litellm_host_python::{Execution, ExecutionBody, ExecutionStep};
use pyo3::{
PyTraverseError, PyVisit,
exceptions::{PyException, PyRuntimeError},

View file

@ -102,7 +102,7 @@ pub(crate) fn call_config(
.without_missing_files(&|path: &Path| path.exists());
let resolution = Resolution::from(&settings);
for unsupported in unreported(&REPORTED_UNSUPPORTED, resolution.unsupported) {
PythonSettings::warn(py, &unsupported.to_string())?;
crate::logger::capture(py).scope(|| litellm_tracing::warn!("{unsupported}"));
}
Ok(resolution.config)
}

View file

@ -4,6 +4,7 @@ mod credentials;
mod diagnostics;
mod errors;
mod http;
mod logger;
mod marshal;
mod python_settings;
mod routes;
@ -25,6 +26,8 @@ mod _native {
#[pymodule_export]
use crate::errors::{RustBridgeDeclined, RustUpstreamError};
#[pymodule_export]
use crate::logger::NativeDiagnosticProcessor;
#[pymodule_export]
use crate::routes::audio_transcription::{atranscription, transcription};
#[pymodule_export]
use crate::routes::chat_completions::{
@ -87,6 +90,7 @@ mod tests {
"chat_completions",
"achat_completions",
"ResponsesWebSocketConnection",
"NativeDiagnosticProcessor",
"TokenCounter",
"Tokenizer",
"gil_stats",

View file

@ -0,0 +1,46 @@
use std::future::Future;
use pyo3::prelude::*;
use serde::Serialize;
pub(crate) fn run_sync<T, E, F>(
py: Python<'_>,
future: F,
map_error: fn(E) -> PyErr,
) -> PyResult<Py<PyAny>>
where
T: Serialize + Send + 'static,
E: Send + 'static,
F: Future<Output = Result<T, E>> + Send + 'static,
{
litellm_host_python::run_sync(py, super::capture(py).instrument(future), map_error)
}
pub(crate) fn run_async<T, E, F>(
py: Python<'_>,
future: F,
map_error: fn(E) -> PyErr,
) -> PyResult<Bound<'_, PyAny>>
where
T: Serialize + Send + 'static,
E: Send + 'static,
F: Future<Output = Result<T, E>> + Send + 'static,
{
litellm_host_python::run_async(py, super::capture(py).instrument(future), map_error)
}
pub(crate) fn run_sync_value<T, F>(py: Python<'_>, future: F) -> PyResult<T>
where
T: Send + 'static,
F: Future<Output = PyResult<T>> + Send + 'static,
{
litellm_host_python::run_sync_value(py, super::capture(py).instrument(future))
}
pub(crate) fn run_async_value<T, F>(py: Python<'_>, future: F) -> PyResult<Bound<'_, PyAny>>
where
T: for<'py> IntoPyObject<'py> + Send + 'static,
F: Future<Output = PyResult<T>> + Send + 'static,
{
litellm_host_python::run_async_value(py, super::capture(py).instrument(future))
}

View file

@ -0,0 +1,41 @@
use std::sync::OnceLock;
use litellm_host::{
host::HostResult,
machine::{HostFailure, Interrupted, Machine, Step},
route::Route,
};
use litellm_tracing::Logger;
use pyo3::Python;
pub(crate) struct LoggedMachine<M> {
machine: M,
logger: OnceLock<Logger>,
}
impl<M> LoggedMachine<M> {
pub(crate) fn new(machine: M) -> Self {
Self {
machine,
logger: OnceLock::new(),
}
}
}
impl<M: Machine> Machine for LoggedMachine<M> {
type Route = M::Route;
type Complete = M::Complete;
fn resume(&mut self, result: Option<HostResult<Self::Route>>) -> Step<'_, Self> {
let logger = self.logger.get_or_init(|| Python::attach(super::capture));
Box::pin(logger.instrument(logger.scope(|| self.machine.resume(result))))
}
fn interrupt(
&mut self,
failure: HostFailure<<Self::Route as Route>::Error>,
) -> Interrupted<'_, Self> {
let logger = self.logger.get_or_init(|| Python::attach(super::capture));
Box::pin(logger.instrument(logger.scope(|| self.machine.interrupt(failure))))
}
}

View file

@ -0,0 +1,166 @@
mod execution;
mod machine;
pub(crate) use execution::{run_async, run_async_value, run_sync, run_sync_value};
pub(crate) use machine::LoggedMachine;
use litellm_host_python::Pythonized;
use litellm_tracing::{DiagnosticInput, Level, Logger, Metadata, Policy, Processor, Record, Sink};
use pyo3::exceptions::PyRuntimeError;
use pyo3::prelude::*;
const MODULE: &str = "litellm.rust_bridge.logger";
type NativeDiagnosticOutput = (String, Option<String>, Option<String>, Vec<String>, bool);
#[pyclass]
pub(crate) struct NativeDiagnosticProcessor {
inner: Processor,
}
#[pymethods]
impl NativeDiagnosticProcessor {
#[new]
fn new(minimum_custom_key_length: usize) -> Self {
Self {
inner: Processor::new(minimum_custom_key_length),
}
}
fn redact_text(&self, text: &str) -> PyResult<String> {
self.inner.redact_text(text).map_err(processing_error)
}
fn redact_structured_text(&self, key: Option<&str>, text: &str) -> PyResult<String> {
self.inner
.redact_structured_text(key, text)
.map_err(processing_error)
}
fn redact_client_message(&self, text: &str) -> PyResult<String> {
self.inner
.redact_client_message(text)
.map_err(processing_error)
}
#[pyo3(signature = (message, exception, stack, leaves, policy))]
fn process_diagnostic(
&self,
message: String,
exception: Option<String>,
stack: Option<String>,
leaves: Vec<(Option<String>, String)>,
policy: (bool, i64, i64),
) -> PyResult<NativeDiagnosticOutput> {
let input = DiagnosticInput {
message,
exception,
stack,
leaves,
};
let policy = Policy {
redact: policy.0,
base64_limit: policy.1,
text_limit: policy.2,
};
self.inner
.process_diagnostic(&input, policy)
.map(|output| {
(
output.message,
output.exception,
output.stack,
output.leaves,
output.changed,
)
})
.map_err(processing_error)
}
fn scrub_access_arguments(&self, arguments: Vec<String>) -> PyResult<Vec<String>> {
self.inner
.scrub_access_arguments(&arguments)
.map_err(processing_error)
}
}
fn processing_error(_: fancy_regex::Error) -> PyErr {
PyRuntimeError::new_err("diagnostic processing failed")
}
struct PythonSink {
correlation: (String, String),
}
fn level(level: &Level) -> u8 {
match *level {
Level::ERROR => 40,
Level::WARN => 30,
Level::INFO => 20,
Level::DEBUG | Level::TRACE => 10,
}
}
fn report<T: Default>(py: Python<'_>, result: PyResult<T>) -> T {
match result {
Ok(value) => value,
Err(error) => {
error.write_unraisable(py, None);
T::default()
}
}
}
impl Sink for PythonSink {
fn enabled(&self, metadata: &Metadata<'_>) -> bool {
if !metadata.target().starts_with("litellm_") && !metadata.target().starts_with("_native::")
{
return false;
}
Python::try_attach(|py| {
report(
py,
py.import(MODULE)
.and_then(|module| module.call_method1("enabled", (level(metadata.level()),)))
.and_then(|enabled| enabled.extract()),
)
})
.unwrap_or(false)
}
fn emit(&self, record: &Record) {
Python::try_attach(|py| {
report(
py,
py.import(MODULE).and_then(|module| {
module
.call_method1(
"emit",
(
level(record.metadata.level()),
&record.message,
record.metadata.file().unwrap_or_default(),
record.metadata.line().unwrap_or_default(),
record.metadata.target(),
Pythonized(&record.fields),
(&self.correlation.0, &self.correlation.1),
),
)
.map(|_| ())
}),
);
});
}
}
pub(crate) fn capture(py: Python<'_>) -> Logger {
report(
py,
py.import(MODULE)
.and_then(|module| module.call_method0("context"))
.and_then(|value| value.extract())
.map(|correlation| Logger::new(PythonSink { correlation })),
)
}
#[cfg(test)]
mod tests;

View file

@ -0,0 +1,295 @@
use std::{process::Command, task::Poll};
use litellm_host::{
host::HostResult,
machine::{HostFailure, Interrupted, Machine, MachineStep, Step},
route::Route,
};
use pyo3::{prelude::*, types::PyDict};
struct DiagnosticMachine;
impl Route for DiagnosticMachine {
type Response = ();
type Error = String;
type Op = ();
type OpResult = ();
type Chunk = ();
type StreamHead = ();
}
impl Machine for DiagnosticMachine {
type Route = Self;
type Complete = ();
fn resume(&mut self, _: Option<HostResult<Self>>) -> Step<'_, Self> {
litellm_tracing::warn!("machine started");
Box::pin(async {
tokio::task::yield_now().await;
litellm_tracing::warn!("machine warning");
Ok(MachineStep::Complete(()))
})
}
fn interrupt(&mut self, _: HostFailure<String>) -> Interrupted<'_, Self> {
Box::pin(async {
litellm_tracing::warn!("machine interrupted");
Ok(())
})
}
}
#[pyfunction]
fn machine_warning(py: Python<'_>) -> PyResult<Bound<'_, PyAny>> {
let mut machine = super::LoggedMachine::new(DiagnosticMachine);
let mut future = Box::pin(async move {
machine
.resume(None)
.await
.map_err(pyo3::exceptions::PyValueError::new_err)?;
machine
.interrupt(HostFailure::Error("stop".into()))
.await
.map_err(pyo3::exceptions::PyValueError::new_err)
});
assert!(matches!(
litellm_host_python::poll_async_value(py, future.as_mut())?,
Poll::Pending
));
litellm_host_python::run_async_value(py, future)
}
#[pyfunction]
fn warning(py: Python<'_>) {
super::capture(py).scope(|| {
litellm_tracing::warn!(attempt = 3, retry = true, "native warning");
});
}
#[pyfunction]
fn levels(py: Python<'_>) {
super::capture(py).scope(|| {
litellm_tracing::trace!("trace");
litellm_tracing::debug!("debug");
litellm_tracing::info!("info");
litellm_tracing::warn!("warn");
litellm_tracing::error!("error");
litellm_tracing::warn!(target: "unrelated_transport", "private wire data");
});
}
#[pyfunction]
fn asynchronous_warning(py: Python<'_>) -> PyResult<Bound<'_, PyAny>> {
super::run_async_value(py, async {
tokio::task::yield_now().await;
litellm_tracing::warn!("async warning");
Ok(())
})
}
#[pyfunction]
fn synchronous_warning(py: Python<'_>) -> PyResult<()> {
super::run_sync_value(py, async {
tokio::task::yield_now().await;
litellm_tracing::warn!("sync warning");
Ok(())
})
}
#[pyfunction]
fn synchronous_failure(py: Python<'_>) -> PyResult<()> {
super::run_sync_value(py, async {
litellm_tracing::warn!("failure diagnostic");
Err(pyo3::exceptions::PyValueError::new_err("request failed"))
})
}
#[pyfunction]
fn http_warning(py: Python<'_>) -> PyResult<()> {
crate::http::call_config(py, &PyDict::new(py), false).map(|_| ())
}
#[test]
fn native_events_reach_python_with_levels_context_reentry_and_http_deduplication() {
if std::env::var_os("LITELLM_LOGGER_TEST_PROCESS").is_none() {
let output = Command::new(std::env::current_exe().unwrap())
.args([
"--exact",
std::thread::current().name().unwrap(),
"--nocapture",
])
.env("LITELLM_LOGGER_TEST_PROCESS", "1")
.output()
.unwrap();
assert!(
output.status.success(),
"{}\n{}",
String::from_utf8_lossy(&output.stdout),
String::from_utf8_lossy(&output.stderr)
);
return;
}
Python::initialize();
Python::attach(|py| {
let locals = PyDict::new(py);
locals
.set_item(
"repo_root",
concat!(env!("CARGO_MANIFEST_DIR"), "/../../.."),
)
.unwrap();
locals
.set_item(
"machine_warning",
wrap_pyfunction!(machine_warning, py).unwrap(),
)
.unwrap();
locals
.set_item(
"synchronous_failure",
wrap_pyfunction!(synchronous_failure, py).unwrap(),
)
.unwrap();
locals
.set_item("levels", wrap_pyfunction!(levels, py).unwrap())
.unwrap();
locals
.set_item("warning", wrap_pyfunction!(warning, py).unwrap())
.unwrap();
locals
.set_item(
"asynchronous_warning",
wrap_pyfunction!(asynchronous_warning, py).unwrap(),
)
.unwrap();
locals
.set_item(
"synchronous_warning",
wrap_pyfunction!(synchronous_warning, py).unwrap(),
)
.unwrap();
locals
.set_item("http_warning", wrap_pyfunction!(http_warning, py).unwrap())
.unwrap();
let importable = py
.eval(
c"__import__('importlib.util', fromlist=['util']).find_spec('dotenv') is not None",
Some(&locals),
Some(&locals),
)
.unwrap()
.is_truthy()
.unwrap();
if !importable {
eprintln!("SKIP: litellm package dependencies are not importable in this interpreter");
return;
}
py.run(c"
import asyncio
import logging
import sys
sys.path.insert(0, repo_root)
import litellm
from litellm._logging import verbose_logger, session_id_var, trace_id_var
class Capture(logging.Handler):
def __init__(self):
super().__init__()
self.records = []
def emit(self, record):
self.records.append(record)
warning()
class Broken(logging.Handler):
def emit(self, record):
raise ValueError('handler failed')
capture = Capture()
old_handlers = verbose_logger.handlers
old_level = verbose_logger.level
old_correlation = litellm.request_correlation_in_logs
old_curve = litellm.ssl_ecdh_curve
old_unraisable = sys.unraisablehook
failures = []
try:
verbose_logger.handlers = [capture]
litellm.request_correlation_in_logs = True
verbose_logger.setLevel(logging.ERROR)
warning()
assert capture.records == []
verbose_logger.setLevel(logging.WARNING)
warning()
assert len(capture.records) == 1
record = capture.records[0]
assert record.getMessage() == 'native warning'
assert record.levelno == logging.WARNING
assert record.rust_fields == {'attempt': 3, 'retry': True}
assert record.pathname.endswith('logger/tests.rs')
assert record.lineno > 0
assert record.rust_target.endswith('logger::tests')
verbose_logger.setLevel(logging.ERROR)
warning()
assert len(capture.records) == 1
verbose_logger.setLevel(logging.WARNING)
async def request(name):
session = session_id_var.set(name)
trace = trace_id_var.set('trace-' + name)
try:
await asynchronous_warning()
await machine_warning()
synchronous_warning()
assert session_id_var.get() == name
assert trace_id_var.get() == 'trace-' + name
finally:
trace_id_var.reset(trace)
session_id_var.reset(session)
async def concurrent():
await asyncio.gather(request('first'), request('second'))
asyncio.run(concurrent())
assert sorted((r.getMessage(), r.session_id, r.trace_id) for r in capture.records[1:]) == sorted(
(message, name, 'trace-' + name)
for name in ('first', 'second')
for message in ('async warning', 'sync warning', 'machine started', 'machine warning', 'machine interrupted')
)
verbose_logger.setLevel(logging.DEBUG)
before_levels = len(capture.records)
levels()
assert [(r.getMessage(), r.levelno) for r in capture.records[before_levels:]] == [
('trace', logging.DEBUG), ('debug', logging.DEBUG), ('info', logging.INFO),
('warn', logging.WARNING), ('error', logging.ERROR),
]
before = len(capture.records)
litellm.ssl_ecdh_curve = 'logger-test-unsupported-curve'
http_warning()
http_warning()
assert len(capture.records) == before + 1
assert 'logger-test-unsupported-curve' in capture.records[-1].getMessage()
assert capture.records[-1].pathname.endswith('http.rs')
verbose_logger.handlers = [Broken()]
sys.unraisablehook = failures.append
warning()
assert len(failures) == 1
assert str(failures[0].exc_value) == 'handler failed'
try:
synchronous_failure()
except ValueError as error:
assert str(error) == 'request failed'
else:
raise AssertionError('request failure was lost')
assert len(failures) == 2
finally:
sys.unraisablehook = old_unraisable
verbose_logger.handlers = old_handlers
verbose_logger.setLevel(old_level)
litellm.request_correlation_in_logs = old_correlation
litellm.ssl_ecdh_curve = old_curve
", Some(&locals), Some(&locals)).unwrap();
});
}

View file

@ -44,11 +44,6 @@ impl PythonSettings {
pub(crate) fn snapshot(self, value: Bound<'_, PyAny>) -> Snapshot<'_> {
Snapshot { group: self, value }
}
pub(crate) fn warn(py: Python<'_>, message: &str) -> PyResult<()> {
py.import(MODULE)?.getattr("warn")?.call1((message,))?;
Ok(())
}
}
#[cfg(test)]

View file

@ -1,7 +1,8 @@
use crate::logger::{run_async, run_sync};
use litellm_core::audio_transcription::{
Error, audio_transcription as run_audio_transcription, types::AudioTranscriptionRequest,
};
use litellm_host_python::{from_py_argument, run_async, run_sync};
use litellm_host_python::from_py_argument;
use pyo3::prelude::*;
use serde_json::{Map, Value};

View file

@ -1,8 +1,9 @@
use crate::logger::{run_async, run_sync};
use litellm_core::chat_completions::{
Error, chat_completions as run_chat_completions, chat_completions_decline_reason,
types::ChatCompletionsRequest,
};
use litellm_host_python::{from_py_argument, run_async, run_sync};
use litellm_host_python::from_py_argument;
use litellm_types::utils::ChatCompletionsResponse;
use pyo3::prelude::*;
use serde_json::{Map, Value};

View file

@ -43,7 +43,7 @@ fn run_messages(
py,
SURFACE,
PublicCall::capture(&request, &args, &kwargs)?,
messages_machine(),
crate::logger::LoggedMachine::new(messages_machine()),
MessagesRouteHost::new(request.unbind()),
asynchronous,
)

View file

@ -73,7 +73,7 @@ fn run_ocr(
py,
if asynchronous { ASYNC_SURFACE } else { SURFACE },
PublicCall::capture(&request, &args, &kwargs)?,
ocr_machine(client),
crate::logger::LoggedMachine::new(ocr_machine(client)),
OcrRouteHost::new(request.unbind()),
asynchronous,
)

View file

@ -25,7 +25,7 @@ impl ResponsesWebSocketConnection {
) -> PyResult<Bound<'py, PyAny>> {
let headers = marshal_headers(headers)?;
let timeout = optional_timeout(timeout_seconds);
litellm_host_python::run_async_value(py, async move {
crate::logger::run_async_value(py, async move {
let inner = RustResponsesWebSocketConnection::connect_url(&url, &headers, timeout)
.await
.map_err(responses_error_to_pyerr)?;
@ -35,7 +35,7 @@ impl ResponsesWebSocketConnection {
fn send_text<'py>(&self, py: Python<'py>, text: String) -> PyResult<Bound<'py, PyAny>> {
let inner = self.inner.clone();
litellm_host_python::run_async_value(py, async move {
crate::logger::run_async_value(py, async move {
inner
.send_text(text)
.await
@ -45,14 +45,14 @@ impl ResponsesWebSocketConnection {
fn recv_text<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
let inner = self.inner.clone();
litellm_host_python::run_async_value(py, async move {
crate::logger::run_async_value(py, async move {
inner.recv_text().await.map_err(responses_error_to_pyerr)
})
}
fn close<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
let inner = self.inner.clone();
litellm_host_python::run_async_value(py, async move {
crate::logger::run_async_value(py, async move {
inner.close().await.map_err(responses_error_to_pyerr)
})
}

View file

@ -1,7 +1,8 @@
use crate::logger::run_async;
use std::sync::Arc;
use std::{num::NonZero, thread::available_parallelism};
use litellm_host_python::{enter_native, run_async};
use litellm_host_python::enter_native;
use litellm_token_counter::{
CountableRequest, Error, InputTokenCount, TokenCounter as CoreTokenCounter,
};

View file

@ -11,7 +11,7 @@ litellm-secrets-types.workspace = true
litellm-core-utils.workspace = true
serde_json.workspace = true
thiserror.workspace = true
tracing = "0.1"
litellm-tracing.workspace = true
veil.workspace = true
aws-sdk-kms = "1.120.0"
aws-sdk-secretsmanager = "1.117.0"

View file

@ -215,7 +215,7 @@ impl AwsSecretsManagerV2 {
.await
.is_err()
{
tracing::warn!("secret created but replication failed");
litellm_tracing::warn!("secret created but replication failed");
}
Ok(response)
}

View file

@ -14,7 +14,7 @@ reqwest.workspace = true
serde_json.workspace = true
thiserror.workspace = true
veil.workspace = true
tracing = "0.1"
litellm-tracing.workspace = true
percent-encoding = "2.3"
tokio = { workspace = true, features = ["sync"] }

View file

@ -92,7 +92,7 @@ impl CyberArkSecretManager {
.unwrap_or(true);
let mut builder = reqwest::Client::builder();
if !verify {
tracing::warn!(
litellm_tracing::warn!(
"CyberArk SSL verification is disabled. This is insecure and should only be used for testing with self-signed certificates."
);
builder = builder.danger_accept_invalid_certs(true);
@ -259,11 +259,13 @@ impl CyberArkSecretManager {
.endpoint
.join(&format!("policies/{}/policy/root", self.account));
let Ok(policy_url) = policy_url else {
tracing::warn!("Could not build CyberArk policy endpoint");
litellm_tracing::warn!("Could not build CyberArk policy endpoint");
return;
};
let Ok(authorization) = self.authorization_header(context).await else {
tracing::warn!("Could not authenticate while ensuring CyberArk variable exists");
litellm_tracing::warn!(
"Could not authenticate while ensuring CyberArk variable exists"
);
return;
};
let body = format!(
@ -288,19 +290,19 @@ impl CyberArkSecretManager {
reqwest::StatusCode::CONFLICT | reqwest::StatusCode::UNPROCESSABLE_ENTITY
) =>
{
tracing::debug!(
litellm_tracing::debug!(
"CyberArk variable policy already exists or conflicts: {}",
response.status()
);
}
Ok(response) => {
tracing::warn!(
litellm_tracing::warn!(
"Could not ensure CyberArk variable exists: {}",
response.status()
);
}
Err(error) => {
tracing::warn!("Error ensuring CyberArk variable exists: {error}");
litellm_tracing::warn!("Error ensuring CyberArk variable exists: {error}");
}
}
}
@ -324,7 +326,7 @@ impl CyberArkSecretManager {
_recovery_window_in_days: Option<u32>,
_context: &SecretOperationContext,
) -> Result<DeleteOutcome, Error> {
tracing::warn!(
litellm_tracing::warn!(
"CyberArk Conjur does not support direct secret deletion. Secrets must be removed through policy updates."
);
self.secrets.invalidate(name).await;

View file

@ -0,0 +1,17 @@
[package]
name = "litellm-tracing"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
fancy-regex.workspace = true
percent-encoding.workspace = true
serde_json.workspace = true
tracing.workspace = true
tracing-subscriber = { version = "0.3", default-features = false, features = ["registry", "std"] }
[dev-dependencies]
rstest.workspace = true
tokio.workspace = true

View file

@ -0,0 +1,26 @@
# Native diagnostic tracing
`litellm-tracing` connects standard `tracing` events to a host-provided `Sink`. It has no Python dependency and does not install a global subscriber
Use the exported `debug!`, `info!`, `warn!`, and `error!` macros in native code. A host creates a `Logger` with its sink, uses `scope` for synchronous operations, and wraps futures with `instrument`. Instrument spawned futures explicitly because thread-local subscribers do not automatically follow spawned work
Bindings implement `litellm_tracing::Sink` to connect events to their host runtime:
```rust
pub trait Sink: Send + Sync + 'static {
fn enabled(&self, metadata: &Metadata<'_>) -> bool;
fn emit(&self, record: &Record);
}
```
Pass the implementation to `litellm_tracing::Logger::new(sink)`, then call `logger.scope(|| litellm_tracing::info!(attempt = 1, "request started"))`. The sink owns host access, level mapping, correlation capture, and delivery failures. `enabled` runs before event fields are evaluated or formatted. `emit` borrows a record; an adapter that queues delivery must copy the data it needs into an owned value
Records retain event metadata, the message, and typed event fields. Sink filtering runs for each event so runtime level changes take effect. Logging from inside a sink is suppressed to prevent recursion
The Python bridge scopes native execution to a sink that uses LiteLLM's existing Python logger. It preserves request correlation, redacts before delivering to handlers, maps Rust trace events to Python debug, and reports handler failures through `sys.unraisablehook`. It accepts LiteLLM targets only, keeping dependency wire diagnostics out of the application logger
Python consumers continue using `litellm._logging` and its existing loggers, filters, formatters, and context setters. Catalog dispatch selects the processing backend for both Python and native diagnostics. The pure `Processor` takes explicit settings and never emits events
A future Node bridge can implement the same sink with runtime-specific delivery and expose the same processor through N-API. Node callback scheduling, queue limits, and shutdown belong in that bridge; this crate has no interpreter handles or output queue
This is diagnostic logging. Request lifecycle hooks and `CustomLogger` dispatch remain separate

View file

@ -0,0 +1,146 @@
use std::{
cell::Cell,
fmt,
future::{Future, poll_fn},
pin::pin,
};
use serde_json::{Map, Value};
use tracing::{
Dispatch, Event, Subscriber,
field::{Field, Visit},
subscriber::Interest,
};
use tracing_subscriber::{Layer, Registry, layer::Context, prelude::*};
mod processing;
mod redaction;
pub use processing::{DiagnosticInput, DiagnosticOutput, Policy, Processor};
pub use redaction::{REDACTED, SecretRedactor};
pub use tracing::{Level, Metadata, debug, error, info, trace, warn};
pub trait Sink: Send + Sync + 'static {
fn enabled(&self, metadata: &Metadata<'_>) -> bool;
fn emit(&self, record: &Record);
}
#[derive(Debug)]
pub struct Record {
pub metadata: &'static Metadata<'static>,
pub message: String,
pub fields: Map<String, Value>,
}
#[derive(Clone, Default)]
pub struct Logger {
dispatch: Dispatch,
}
impl Logger {
pub fn new(sink: impl Sink) -> Self {
Self {
dispatch: Dispatch::new(Registry::default().with(Output(sink))),
}
}
pub fn scope<T>(&self, operation: impl FnOnce() -> T) -> T {
if EMITTING.get() {
return operation();
}
tracing::dispatcher::with_default(&self.dispatch, operation)
}
pub fn instrument<F: Future>(&self, future: F) -> impl Future<Output = F::Output> + use<F> {
let logger = self.clone();
async move {
let mut future = pin!(future);
poll_fn(|context| logger.scope(|| future.as_mut().poll(context))).await
}
}
}
thread_local! {
static EMITTING: Cell<bool> = const { Cell::new(false) };
}
struct Emitting;
impl Emitting {
fn enter() -> Option<Self> {
EMITTING.with(|active| (!active.replace(true)).then_some(Self))
}
}
impl Drop for Emitting {
fn drop(&mut self) {
EMITTING.set(false);
}
}
struct Output<S>(S);
impl<S: Sink, R: Subscriber> Layer<R> for Output<S> {
fn register_callsite(&self, _: &'static Metadata<'static>) -> Interest {
Interest::sometimes()
}
fn enabled(&self, metadata: &Metadata<'_>, _: Context<'_, R>) -> bool {
let Some(_guard) = Emitting::enter() else {
return false;
};
self.0.enabled(metadata)
}
fn on_event(&self, event: &Event<'_>, _: Context<'_, R>) {
let Some(_guard) = Emitting::enter() else {
return;
};
let mut record = Record {
metadata: event.metadata(),
message: String::new(),
fields: Map::new(),
};
event.record(&mut record);
self.0.emit(&record);
}
}
impl Record {
fn field(&mut self, field: &Field, value: Value) {
if field.name() == "message" {
self.message = match value {
Value::String(message) => message,
value => value.to_string(),
};
} else {
self.fields.insert(field.name().to_owned(), value);
}
}
}
impl Visit for Record {
fn record_debug(&mut self, field: &Field, value: &dyn fmt::Debug) {
self.field(field, format!("{value:?}").into());
}
fn record_str(&mut self, field: &Field, value: &str) {
self.field(field, value.into());
}
fn record_bool(&mut self, field: &Field, value: bool) {
self.field(field, value.into());
}
fn record_i64(&mut self, field: &Field, value: i64) {
self.field(field, value.into());
}
fn record_u64(&mut self, field: &Field, value: u64) {
self.field(field, value.into());
}
fn record_f64(&mut self, field: &Field, value: f64) {
self.field(field, value.into());
}
}

View file

@ -0,0 +1,333 @@
use fancy_regex::Result;
use percent_encoding::percent_decode_str;
use crate::{REDACTED, SecretRedactor};
#[derive(Clone, Copy, Debug)]
pub struct Policy {
pub redact: bool,
pub base64_limit: i64,
pub text_limit: i64,
}
#[derive(Clone, Debug)]
pub struct DiagnosticInput {
pub message: String,
pub exception: Option<String>,
pub stack: Option<String>,
pub leaves: Vec<(Option<String>, String)>,
}
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct DiagnosticOutput {
pub message: String,
pub exception: Option<String>,
pub stack: Option<String>,
pub leaves: Vec<String>,
pub changed: bool,
}
pub struct Processor {
redactor: SecretRedactor,
}
impl Processor {
pub fn new(minimum_custom_key_length: usize) -> Self {
Self {
redactor: SecretRedactor::new(minimum_custom_key_length),
}
}
pub fn redact_text(&self, text: &str) -> Result<String> {
self.redactor.try_redact(text)
}
pub fn redact_structured_text(&self, key: Option<&str>, text: &str) -> Result<String> {
self.redactor.try_redact_structured(key, text)
}
pub fn redact_client_message(&self, text: &str) -> Result<String> {
self.redactor.try_redact_internal(text)
}
pub fn process_diagnostic(
&self,
input: &DiagnosticInput,
policy: Policy,
) -> Result<DiagnosticOutput> {
let message = self.process_text(&input.message, policy)?;
let exception = input
.exception
.as_deref()
.map(|text| self.process_text(text, policy))
.transpose()?;
let stack = input
.stack
.as_deref()
.map(|text| {
if policy.redact {
self.redact_text(text)
} else {
Ok(text.to_owned())
}
})
.transpose()?;
let leaves = input
.leaves
.iter()
.map(|(key, text)| {
if policy.redact {
self.redact_structured_text(key.as_deref(), text)
} else {
Ok(text.clone())
}
})
.collect::<Result<Vec<_>>>()?;
let changed = message != input.message
|| exception != input.exception
|| stack != input.stack
|| leaves
.iter()
.zip(&input.leaves)
.any(|(processed, (_, original))| processed != original);
Ok(DiagnosticOutput {
message,
exception,
stack,
leaves,
changed,
})
}
pub fn scrub_access_arguments(&self, arguments: &[String]) -> Result<Vec<String>> {
arguments
.iter()
.map(|argument| self.scrub_access_arg(argument))
.collect()
}
fn process_text(&self, text: &str, policy: Policy) -> Result<String> {
let collapsed = if policy.base64_limit > 0 {
collapse_base64(text, policy.base64_limit as usize)
} else {
text.to_owned()
};
let redacted = if policy.redact {
self.redact_text(&collapsed)?
} else {
collapsed
};
Ok(
if policy.text_limit > 0 && redacted.chars().count() > policy.text_limit as usize {
truncate_text(&redacted, policy.text_limit as usize)
} else {
redacted
},
)
}
fn scrub_access_arg(&self, value: &str) -> Result<String> {
let length = value.chars().count();
let scanned = if length <= 512 {
value
} else {
let head = &value[..char_offset(value, 512)];
if head.contains('?') {
&head[..head.rfind(['?', '&']).unwrap_or(0)]
} else {
head
}
};
let scrubbed = self.redact_text(scanned)?;
let (path, query) = scrubbed
.split_once('?')
.map_or((scrubbed.as_str(), None), |(path, query)| {
(path, Some(query))
});
let safe = if self.hides_encoded_credential(path)? {
REDACTED.to_owned()
} else if query.is_some() && self.hides_encoded_credential(&scrubbed)? {
format!("{path}?{REDACTED}")
} else {
scrubbed
};
Ok(if length > 512 {
format!(
"{safe}... ({} more chars truncated) ...",
length - scanned.chars().count()
)
} else {
safe
})
}
fn hides_encoded_credential(&self, value: &str) -> Result<bool> {
if !value.as_bytes().contains(&b'%') {
return Ok(false);
}
let decoded = percent_decode_str(value).decode_utf8_lossy();
Ok(self.redact_text(&decoded)? != decoded)
}
}
fn char_offset(text: &str, count: usize) -> usize {
text.char_indices()
.nth(count)
.map_or(text.len(), |(index, _)| index)
}
fn marker(skipped_chars: usize) -> String {
format!(
"... (litellm_truncated skipped {skipped_chars} chars. Truncation is a stdout logging safeguard. Full, untruncated data is logged to logging callbacks (OTEL, Datadog, etc.) and at DEBUG level. To increase the truncation limit, set `MAX_STRING_LENGTH_STDOUT_LOG` in your env.) ..."
)
}
fn truncate_text(text: &str, limit: usize) -> String {
let length = text.chars().count();
let kept = limit.saturating_sub(marker(length).len());
if kept == 0 {
return text[..char_offset(text, limit)].to_owned();
}
let head = kept / 2;
let tail = kept - head;
format!(
"{}{}{}",
&text[..char_offset(text, head)],
marker(length - kept),
&text[char_offset(text, length - tail)..]
)
}
fn base64_byte(byte: u8) -> bool {
byte.is_ascii_alphanumeric() || byte == b'+' || byte == b'/'
}
fn looks_like_base64(run: &str) -> bool {
let unpadded = run.trim_end_matches('=');
let lower_hex = unpadded
.bytes()
.all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte));
let upper_hex = unpadded
.bytes()
.all(|byte| byte.is_ascii_digit() || (b'A'..=b'F').contains(&byte));
let repeated = unpadded.bytes().all(|byte| byte == unpadded.as_bytes()[0]);
(!lower_hex && !upper_hex) || repeated
}
fn base64_size(chars: usize) -> String {
let bytes = chars as f64 * 3.0 / 4.0;
if bytes >= 1024.0 * 1024.0 {
return format!("{:.2}MB", bytes / (1024.0 * 1024.0));
}
if bytes >= 1024.0 {
return format!("{:.1}KB", bytes / 1024.0);
}
format!("{}B", bytes as usize)
}
fn collapse_base64(text: &str, limit: usize) -> String {
let bytes = text.as_bytes();
let mut position = 0;
let mut previous = 0;
let mut output = String::new();
while position < bytes.len() {
if !base64_byte(bytes[position]) || (position > 0 && base64_byte(bytes[position - 1])) {
position += 1;
continue;
}
let start = position;
while position < bytes.len() && base64_byte(bytes[position]) {
position += 1;
}
let run_end = position;
while position < bytes.len() && position - run_end < 2 && bytes[position] == b'=' {
position += 1;
}
let run = &text[start..position];
if run_end - start > limit && looks_like_base64(run) {
output.push_str(&text[previous..start]);
output.push_str(&format!(
"[base64_data truncated: {}]",
base64_size(run.len())
));
previous = position;
}
}
output.push_str(&text[previous..]);
output
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn redaction_precedes_the_text_bound_and_preserves_unicode_character_limits() {
let processor = Processor::new(16);
let secret = format!("sk-{}", "q".repeat(48));
let text = format!("{}{}{}", "é".repeat(110), secret, "界".repeat(1000));
let input = DiagnosticInput {
message: text,
exception: None,
stack: None,
leaves: vec![],
};
let output = processor
.process_diagnostic(
&input,
Policy {
redact: true,
base64_limit: 0,
text_limit: 500,
},
)
.unwrap();
assert!(output.message.chars().count() <= 500);
assert!(!output.message.contains("sk-qq"));
assert!(output.changed);
}
#[test]
fn base64_collapse_applies_to_debug_and_exceptions_without_touching_hex() {
let processor = Processor::new(16);
let input = DiagnosticInput {
message: format!("image={} digest={}", "Q".repeat(100), "a1".repeat(50)),
exception: Some(format!("upload failed: {}", "Q".repeat(100))),
stack: Some("api_key=secret123".to_owned()),
leaves: vec![(Some("api_key".to_owned()), "secret123".to_owned())],
};
let output = processor
.process_diagnostic(
&input,
Policy {
redact: true,
base64_limit: 20,
text_limit: 0,
},
)
.unwrap();
assert!(output.message.contains("[base64_data truncated: 75B]"));
assert!(output.message.contains(&"a1".repeat(50)));
assert!(
output
.exception
.unwrap()
.contains("[base64_data truncated: 75B]")
);
assert_eq!(output.stack.as_deref(), Some(REDACTED));
assert_eq!(output.leaves, vec![REDACTED]);
}
#[test]
fn access_arguments_keep_encoded_paths_and_drop_decoded_credentials() {
let processor = Processor::new(16);
let arguments = vec![
"/v1/models?filter=gpt%2D4o&page=2".to_owned(),
"/v1/models?k%65y=sk%2Dabcdefghijklmnopqrstuvwxyz&page=2".to_owned(),
];
assert_eq!(
processor.scrub_access_arguments(&arguments).unwrap(),
vec![arguments[0].clone(), "/v1/models?REDACTED".to_owned()]
);
}
}

View file

@ -0,0 +1,140 @@
use fancy_regex::{NoExpand, Regex};
pub const REDACTED: &str = "REDACTED";
#[cfg(test)]
const DEFAULT_MINIMUM_CUSTOM_KEY_LENGTH: usize = 16;
fn secret_patterns(minimum_custom_key_length: usize) -> String {
let sk_suffix_length = minimum_custom_key_length.saturating_sub("sk-".len());
[
r"-----BEGIN[A-Z \-]*PRIVATE KEY-----[\s\S]*?-----END[A-Z \-]*PRIVATE KEY-----",
r"\bya29\.[A-Za-z0-9_.~+/-]+",
r#"(?:client_secret|azure_password|azure_username)\s+[^\s,'"})\]{}>]+"#,
r"(?:AKIA|ASIA)[0-9A-Z]{16}",
r"Bearer\s+[A-Za-z0-9\-._~+/]{10,}=*",
r"Basic\s+[A-Za-z0-9+/]{10,}={0,2}",
&format!(r"sk-[A-Za-z0-9\-_]{{{sk_suffix_length},}}"),
r#"(?<=[?&])(?:api[_-]?key|\w*(?:token|password|passwd|client_secret|secret_key|_secret))=[^\s&'"]+"#,
r#"(?:api[_-]?key)['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]{8,}"#,
r#"(?:x-api-key|api-key)['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]+"#,
r"x-ak-[A-Za-z0-9\-_]{20,}",
r"AIza[0-9A-Za-z\-_]{35}",
r#"(?<=[?&])key=[^\s&'"]{8,}"#,
r#"(?:^|(?<=\W))\w*(?:password|passwd|client_secret|secret_key|_secret)['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]+"#,
r#"(?<=://)[^\s'":]{0,4096}:[^\s'"]{1,4096}(?=@)"#,
r"dapi[0-9a-f]{32}",
r#"litellm\.[A-Za-z0-9_]*_key['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]+"#,
r#"private_key['"]?\s*[:=]\s*['"]?(?:-----BEGIN[A-Z \-]*PRIVATE KEY-----[\s\S]*?-----END[A-Z \-]*PRIVATE KEY-----|[^\s,'"})\]{}>]+)"#,
concat!(
r"(?:master_key|xai_key|database_url|db_url|connection_string|",
r"aws_secret_access_key|aws_session_token|aws_access_key_id|s3_secret_access_key|s3_access_key_id|",
r"signing_key|encryption_key|",
r"auth_token|access_token|refresh_token|",
r"slack_webhook_url|webhook_url|",
r"database_connection_string|",
r"huggingface_token|jwt_secret)",
r#"['"]?\s*[:=]\s*['"]?[^\s,'"})\]{}>]+"#,
),
r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]*",
r"(?<=[?&])sig=[A-Za-z0-9%+/=]+",
r#"\{[^{}]*"type"\s*:\s*"service_account"[^{}]*(?:\{[^{}]*\}[^{}]*)*\}"#,
]
.join("|")
}
#[derive(Clone, Debug)]
pub struct SecretRedactor {
pattern: Regex,
internal_pattern: Regex,
}
impl SecretRedactor {
pub fn new(minimum_custom_key_length: usize) -> Self {
let pattern = Regex::new(&format!(
"(?i){}",
secret_patterns(minimum_custom_key_length)
))
.expect("secret redaction patterns compile");
let internal_pattern = Regex::new(concat!(
r#"(?i)/(?:etc|var|opt|usr|home|root|private|Users|tmp|mnt|srv)/[^\s'"\)\]}>,]+|"#,
r#"[A-Za-z]:\\[^\s'"\)\]}>,]+|"#,
r"\b(?:10(?:\.\d{1,3}){3}|172\.(?:1[6-9]|2\d|3[01])(?:\.\d{1,3}){2}|",
r"192\.168(?:\.\d{1,3}){2}|127(?:\.\d{1,3}){3})\b|",
r"\b[A-Za-z0-9-]+(?:\.[A-Za-z0-9-]+)*\.(?:internal|local|corp|lan|intra|private)\b",
))
.expect("internal detail patterns compile");
Self {
pattern,
internal_pattern,
}
}
pub fn redact(&self, value: &str) -> String {
self.try_redact(value)
.unwrap_or_else(|_| REDACTED.to_owned())
}
pub fn try_redact(&self, value: &str) -> fancy_regex::Result<String> {
self.pattern
.try_replacen(value, 0, NoExpand(REDACTED))
.map(|value| value.into_owned())
}
pub fn try_redact_structured(
&self,
key: Option<&str>,
value: &str,
) -> fancy_regex::Result<String> {
let scrubbed = self.try_redact(value)?;
if scrubbed != value || key.is_none() {
return Ok(scrubbed);
}
let rendered = format!("'{}': '{value}'", key.unwrap_or_default());
Ok(if self.try_redact(&rendered)? != rendered {
REDACTED.to_owned()
} else {
value.to_owned()
})
}
pub fn try_redact_internal(&self, value: &str) -> fancy_regex::Result<String> {
let without_traceback = value
.split_once("Traceback (most recent call last):")
.map_or(value, |(prefix, _)| prefix.trim_end());
self.internal_pattern
.try_replacen(&self.try_redact(without_traceback)?, 0, NoExpand(REDACTED))
.map(|value| value.into_owned())
}
}
#[cfg(test)]
mod tests {
use super::*;
#[rstest::rstest]
#[case::bearer("auth failed: Bearer abcdefghijklmnop", "auth failed: REDACTED")]
#[case::sk_key("key sk-abcdefghijklmnopqrstuvwxyz rejected", "key REDACTED rejected")]
#[case::short_sk_key_is_kept("sk-abc", "sk-abc")]
#[case::query_param("GET /v1?api_key=secret123&x=1", "GET /v1?REDACTED&x=1")]
#[case::dict_repr("{'api_key': 'abcdefghij'}", "{'REDACTED'}")]
#[case::url_credentials("postgres://user:pass@host/db", "postgres://REDACTED@host/db")]
#[case::case_insensitive("BEARER ABCDEFGHIJKLMNOP", "REDACTED")]
#[case::aws_key("AKIAABCDEFGHIJKLMNOP", "REDACTED")]
#[case::sas_signature("https://x.blob/a?sv=1&sig=abc%2B=", "https://x.blob/a?sv=1&REDACTED")]
#[case::password_needs_word_boundary("db_password=hunter2", "REDACTED")]
#[case::plain_text_is_kept(r#"{"message": "rejected"}"#, r#"{"message": "rejected"}"#)]
fn redacts_the_same_spans_as_the_python_patterns(#[case] input: &str, #[case] expected: &str) {
assert_eq!(
SecretRedactor::new(DEFAULT_MINIMUM_CUSTOM_KEY_LENGTH).redact(input),
expected
);
}
#[test]
fn sk_threshold_follows_the_minimum_custom_key_length() {
let redactor = SecretRedactor::new(8);
assert_eq!(redactor.redact("sk-abcde"), REDACTED);
assert_eq!(redactor.redact("sk-abcd"), "sk-abcd");
}
}

View file

@ -0,0 +1,122 @@
use std::sync::{
Arc,
atomic::{AtomicBool, Ordering},
mpsc,
};
use litellm_tracing::{Level, Logger, Metadata, Record, Sink, info, warn};
use serde_json::{Value, json};
struct Output {
enabled: Arc<AtomicBool>,
sender: mpsc::Sender<(String, Value, Level, &'static str, Option<u32>)>,
}
impl Sink for Output {
fn enabled(&self, _: &Metadata<'_>) -> bool {
self.enabled.load(Ordering::Relaxed)
}
fn emit(&self, record: &Record) {
self.sender
.send((
record.message.clone(),
Value::Object(record.fields.clone()),
*record.metadata.level(),
record.metadata.target(),
record.metadata.line(),
))
.unwrap();
Logger::default().scope(|| warn!("a sink must not recursively emit"));
}
}
fn emit() {
warn!(
attempt = 3_u64,
elapsed = 1.5,
retry = true,
reason = "timeout",
"retry {}",
3
);
}
#[test]
fn records_preserve_fields_metadata_and_dynamic_filtering_without_recursion() {
let (sender, receiver) = mpsc::channel();
let enabled = Arc::new(AtomicBool::new(false));
let logger = Logger::new(Output {
enabled: enabled.clone(),
sender,
});
logger.scope(emit);
assert!(receiver.try_recv().is_err());
enabled.store(true, Ordering::Relaxed);
logger.scope(emit);
let (message, fields, level, target, line) = receiver.try_recv().unwrap();
assert_eq!(message, "retry 3");
assert_eq!(
fields,
json!({"attempt": 3, "elapsed": 1.5, "retry": true, "reason": "timeout"})
);
assert_eq!(level, Level::WARN);
assert_eq!(target, module_path!());
assert!(line.is_some());
enabled.store(false, Ordering::Relaxed);
logger.scope(emit);
assert!(receiver.try_recv().is_err());
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn concurrent_futures_keep_their_sinks_across_suspension_and_spawn() {
let tasks = (0..2)
.map(|id| {
let (sender, receiver) = mpsc::channel();
let logger = Logger::new(Output {
enabled: Arc::new(AtomicBool::new(true)),
sender,
});
let task = tokio::spawn(logger.instrument(async move {
tokio::task::yield_now().await;
info!(id, "worker");
}));
(id, task, receiver)
})
.collect::<Vec<_>>();
for (id, task, receiver) in tasks {
task.await.unwrap();
let (message, fields, level, _, _) = receiver.try_recv().unwrap();
assert_eq!(message, "worker");
assert_eq!(fields, json!({"id": id}));
assert_eq!(level, Level::INFO);
assert!(receiver.try_recv().is_err());
}
}
#[test]
fn nested_scopes_restore_the_previous_sink() {
let (outer_sender, outer) = mpsc::channel();
let (inner_sender, inner) = mpsc::channel();
let logger = |sender| {
Logger::new(Output {
enabled: Arc::new(AtomicBool::new(true)),
sender,
})
};
let outside = logger(outer_sender);
let inside = logger(inner_sender);
outside.scope(|| {
info!("before");
inside.scope(|| info!("inside"));
info!("after");
});
assert_eq!(
outer.try_iter().map(|event| event.0).collect::<Vec<_>>(),
["before", "after"]
);
assert_eq!(
inner.try_iter().map(|event| event.0).collect::<Vec<_>>(),
["inside"]
);
}

View file

@ -1,11 +1,12 @@
import ast
import contextvars
import functools
import itertools
import logging
import os
import re
import sys
from collections.abc import Sequence
from collections.abc import Iterator
from datetime import datetime
from logging import Formatter
from typing import Any, Final, TextIO
@ -22,10 +23,12 @@ from litellm.litellm_core_utils.env_utils import get_env_int
from litellm.litellm_core_utils.safe_json_dumps import UNSERIALIZABLE_OBJECT, safe_dumps, safe_json_structure
from litellm.litellm_core_utils.safe_json_loads import safe_json_loads
from litellm.litellm_core_utils.secret_redaction import (
_python_redact_string,
_python_redact_structured_value,
redact_internal_details,
redact_string,
redact_structured_value,
)
from litellm.rust_bridge import diagnostics
set_verbose = False
@ -49,7 +52,7 @@ def _sanitize_correlation_id(value: str) -> str:
pass through credential redaction.
"""
stripped: Final = "".join(ch for ch in value if ch.isprintable())
return _redact_string(stripped[:_MAX_CORRELATION_ID_LENGTH])
return _redact_string(stripped)[:_MAX_CORRELATION_ID_LENGTH]
def set_session_id(session_id: str) -> "contextvars.Token[str]":
@ -74,12 +77,6 @@ def _redact_string(value: str) -> str:
return redact_string(value)
def _redact_structured_value(key: str | None, value: str) -> str:
if not _ENABLE_SECRET_REDACTION:
return value
return redact_structured_value(key, value)
_REDACTED_RECORD_ATTR: Final = "litellm_redacted"
_REDACTED_STAMP: Final = object()
_UNREDACTED_SCALAR_TYPES: Final = (bool, int, float, type(None))
@ -103,14 +100,6 @@ def _plain_text(value: object) -> str:
return UNSERIALIZABLE_OBJECT
def _redact_extra_value(key: str, value: object) -> object:
try:
scrubbed: Final = safe_json_structure(value, value_transform=_redact_structured_value, key=key)
except Exception:
return _redact_string(_plain_text(value))
return value if _scrubbing_changed_nothing(scrubbed, value) else scrubbed
def redact_secrets(value: str) -> str:
"""Public API: redact known secret/credential patterns from an arbitrary string.
@ -150,7 +139,7 @@ def _substituted_color_message(record: logging.LogRecord) -> str | None:
return None
try:
return color_message % record.args
except TypeError:
except Exception:
return color_message
@ -162,42 +151,7 @@ class SecretRedactionFilter(logging.Filter):
def filter(self, record: logging.LogRecord) -> bool:
if not _ENABLE_SECRET_REDACTION or _is_redacted(record):
return True
# Runs before args are cleared, and before the extra-field loop below
# that redacts the substituted result.
substituted_color_message: Final = _substituted_color_message(record)
if substituted_color_message is not None:
record.color_message = substituted_color_message # rebind-ok: a Filter scrubs records in place
try:
record.msg = _redact_string(record.getMessage())
record.args = None
except Exception:
if isinstance(record.msg, str):
record.msg = _redact_string(record.msg)
# Redact exception tracebacks
if record.exc_info and record.exc_info[1] is not None:
try:
record.exc_text = _redact_string(record.exc_text or self._formatter.formatException(record.exc_info))
except Exception:
pass
if isinstance(record.stack_info, str):
record.stack_info = _redact_string(record.stack_info) # rebind-ok: a Filter scrubs records in place
# Redact extra fields passed via logger.debug("msg", extra={...})
record_items: Final[Sequence[tuple[str, object]]] = list(record.__dict__.items())
for key, value in record_items:
if key in _STANDARD_RECORD_ATTRS:
continue
if isinstance(value, str):
setattr(record, key, _redact_structured_value(key, value))
elif not isinstance(value, _UNREDACTED_SCALAR_TYPES):
setattr(record, key, _redact_extra_value(key, value))
setattr(record, _REDACTED_RECORD_ATTR, _REDACTED_STAMP)
return True
return _process_record(record, base64_limit=0, text_limit=0, redact=True)
_secret_filter: Final = SecretRedactionFilter()
@ -211,7 +165,7 @@ _REDACTION_PLACEHOLDER: Final = "REDACTED"
def _hides_a_credential(value: str) -> bool:
"""Whether *value* only looks clean until it is percent-decoded."""
decoded: Final = unquote(value)
return _redact_string(decoded) != decoded
return _python_redact_string(decoded) != decoded
def _drop_encoded_credential(scrubbed: str) -> str:
@ -239,10 +193,10 @@ def _scrub_access_arg(value: str) -> str:
pattern and would then be logged raw.
"""
if len(value) <= _MAX_SCRUBBED_ACCESS_ARG:
return _drop_encoded_credential(_redact_string(value))
return _drop_encoded_credential(_python_redact_string(value))
head: Final = value[:_MAX_SCRUBBED_ACCESS_ARG]
kept: Final = head[: max(head.rfind("?"), head.rfind("&"))] if "?" in head else head
scrubbed: Final = _drop_encoded_credential(_redact_string(kept))
scrubbed: Final = _drop_encoded_credential(_python_redact_string(kept))
return f"{scrubbed}... ({len(value) - len(kept)} more chars truncated) ..."
@ -258,8 +212,17 @@ class AccessLogRedactionFilter(logging.Filter):
if not _ENABLE_SECRET_REDACTION:
return True
if isinstance(record.args, tuple) and record.args:
strings: Final = tuple(arg for arg in record.args if isinstance(arg, str))
candidate: Final = diagnostics.run(
lambda native: native.scrub_access_arguments(strings),
lambda: tuple(_scrub_access_arg(arg) for arg in strings),
)
scrubbed: Final = (
candidate if len(candidate) == len(strings) else tuple(_scrub_access_arg(arg) for arg in strings)
)
values: Final = iter(scrubbed)
record.args = tuple( # rebind-ok: a Filter scrubs records in place
_scrub_access_arg(arg) if isinstance(arg, str) else arg for arg in record.args
next(values) if isinstance(arg, str) else arg for arg in record.args
)
return True
# No positional args means everything is in msg, where collapsing is correct.
@ -365,6 +328,185 @@ def _collapse_base64_runs(text: str, limit: int) -> str:
return _base64_run_pattern(limit + 1).sub(_replace_base64_run, text)
def _extra_structure(key: str, value: object) -> object:
if isinstance(value, str):
return value
try:
return safe_json_structure(value, key=key)
except Exception:
return _plain_text(value)
def _string_leaves(key: str | None, value: object) -> Iterator[tuple[str | None, str]]:
if isinstance(value, str):
yield key, value
elif isinstance(value, dict):
yield from itertools.chain.from_iterable(
_string_leaves(child_key, child) for child_key, child in value.items() if isinstance(child_key, str)
)
elif isinstance(value, (list, tuple)):
yield from itertools.chain.from_iterable(_string_leaves(key, child) for child in value)
def _replace_string_leaves(value: object, values: Iterator[str]) -> object:
if isinstance(value, str):
return next(values)
if isinstance(value, dict):
return { # mutable-ok: LogRecord extras must keep JSON dict shape for handlers
key: _replace_string_leaves(child, values) for key, child in value.items()
}
if isinstance(value, list):
return [ # mutable-ok: LogRecord extras must keep JSON list shape for handlers
_replace_string_leaves(child, values) for child in value
]
if isinstance(value, tuple):
return tuple(_replace_string_leaves(child, values) for child in value)
return value
def _sort_processed_sets(original: object, processed: object) -> object:
if isinstance(original, set) and isinstance(processed, list):
return sorted(processed)
if isinstance(original, dict) and isinstance(processed, dict):
return { # mutable-ok: sorting nested sets must preserve the surrounding JSON dict
key: _sort_processed_sets(original.get(key), value) for key, value in processed.items()
}
if isinstance(original, list) and isinstance(processed, list):
return [ # mutable-ok: sorting nested sets must preserve the surrounding JSON list
_sort_processed_sets(before, after) for before, after in zip(original, processed)
]
if isinstance(original, tuple) and isinstance(processed, tuple):
return tuple(_sort_processed_sets(before, after) for before, after in zip(original, processed))
return processed
def _python_process_diagnostic(
message: str,
exception: str | None,
stack: str | None,
leaves: tuple[tuple[str | None, str], ...],
redact: bool,
base64_limit: int,
text_limit: int,
) -> tuple[str, str | None, str | None, tuple[str, ...], bool]:
def process_text(text: str) -> str:
collapsed: Final = _collapse_base64_runs(text, base64_limit) if base64_limit > 0 else text
scrubbed: Final = _python_redact_string(collapsed) if redact else collapsed
return _truncate_for_stdout_log(scrubbed, text_limit) if 0 < text_limit < len(scrubbed) else scrubbed
processed_message: Final = process_text(message)
processed_exception: Final = process_text(exception) if exception is not None else None
processed_stack: Final = _python_redact_string(stack) if redact and stack is not None else stack
processed_leaves: Final = tuple(
_python_redact_structured_value(key, text) if redact else text for key, text in leaves
)
changed: Final = (
processed_message != message
or processed_exception != exception
or processed_stack != stack
or any(processed != original for processed, (_, original) in zip(processed_leaves, leaves))
)
return processed_message, processed_exception, processed_stack, processed_leaves, changed
def _render_message(record: logging.LogRecord) -> str:
try:
return record.getMessage()
except Exception:
return record.msg if isinstance(record.msg, str) else UNSERIALIZABLE_OBJECT
def _render_exception(record: logging.LogRecord) -> str | None:
if not isinstance(record.exc_info, tuple) or len(record.exc_info) < 2 or record.exc_info[1] is None:
return None
try:
return record.exc_text or SecretRedactionFilter._formatter.formatException(record.exc_info)
except Exception:
return "REDACTED"
def _process_record(record: logging.LogRecord, *, base64_limit: int, text_limit: int, redact: bool) -> bool:
if _is_redacted(record):
return True
message: Final = _render_message(record)
exception: Final = _render_exception(record)
stack: Final = record.stack_info if isinstance(record.stack_info, str) else None
substituted_color: Final = _substituted_color_message(record)
extras: Final = (
tuple(
(
key,
value,
_extra_structure(
key, substituted_color if key == "color_message" and substituted_color is not None else value
),
)
for key, value in record.__dict__.items()
if key not in _STANDARD_RECORD_ATTRS
and key != _REDACTED_RECORD_ATTR
and not isinstance(value, _UNREDACTED_SCALAR_TYPES)
)
if redact
else ()
)
extra_leaves: Final = tuple(
itertools.chain.from_iterable(_string_leaves(key, prepared) for key, _, prepared in extras)
)
raw_template: Final = record.msg if redact and isinstance(record.msg, str) and record.args else None
color_template: Final = record.__dict__.get("color_message")
raw_color: Final = color_template if redact and isinstance(color_template, str) and record.args else None
leaves: Final = (
extra_leaves
+ (((None, raw_template),) if raw_template is not None else ())
+ (((None, raw_color),) if raw_color is not None else ())
)
candidate: Final = diagnostics.run(
lambda native: native.process_diagnostic(message, exception, stack, leaves, (redact, base64_limit, text_limit)),
lambda: _python_process_diagnostic(message, exception, stack, leaves, redact, base64_limit, text_limit),
)
processed_message, processed_exception, processed_stack, processed_leaves, _ = (
candidate
if len(candidate[3]) == len(leaves)
else _python_process_diagnostic(message, exception, stack, leaves, redact, base64_limit, text_limit)
)
raw_template_changed: Final = raw_template is not None and processed_leaves[len(extra_leaves)] != raw_template
safe_message: Final = "REDACTED" if raw_template_changed and processed_message == message else processed_message
if redact or safe_message != message:
record.msg = safe_message # rebind-ok: the Filter interface mutates the record
record.args = None # rebind-ok: the rendered message replaces interpolation inputs
if processed_exception is not None:
record.exc_text = processed_exception # rebind-ok: the Filter interface mutates the record
if processed_stack is not None:
record.stack_info = processed_stack # rebind-ok: the Filter interface mutates the record
processed_values: Final = iter(processed_leaves[: len(extra_leaves)])
for key, original, prepared in extras:
replacement: Final = _sort_processed_sets(original, _replace_string_leaves(prepared, processed_values))
if not _scrubbing_changed_nothing(replacement, original):
setattr(record, key, replacement)
raw_color_changed: Final = (
raw_color is not None and processed_leaves[len(extra_leaves) + int(raw_template is not None)] != raw_color
)
if raw_color_changed and getattr(record, "color_message", None) == substituted_color:
setattr(record, "color_message", "REDACTED")
setattr(record, _REDACTED_RECORD_ATTR, _REDACTED_STAMP)
return True
def _redact_json_record(value: object) -> object:
prepared: Final = safe_json_structure(value)
leaves: Final = tuple(_string_leaves(None, prepared))
candidate: Final = diagnostics.run(
lambda native: native.process_diagnostic("", None, None, leaves, (True, 0, 0))[3],
lambda: tuple(_python_redact_structured_value(key, text) for key, text in leaves),
)
replacements: Final = (
candidate
if len(candidate) == len(leaves)
else tuple(_python_redact_structured_value(key, text) for key, text in leaves)
)
return _sort_processed_sets(value, _replace_string_leaves(prepared, iter(replacements)))
class StdoutLogTruncationFilter(logging.Filter):
"""Bounds how much of an oversized log line reaches stdout.
@ -412,7 +554,17 @@ class StdoutLogTruncationFilter(logging.Filter):
return True
_stdout_truncation_filter: Final = StdoutLogTruncationFilter()
class DiagnosticProcessingFilter(StdoutLogTruncationFilter):
def filter(self, record: logging.LogRecord) -> bool:
return _process_record(
record,
base64_limit=_get_max_base64_length_stdout_log(),
text_limit=_get_max_string_length_stdout_log() if record.levelno >= logging.INFO else 0,
redact=_ENABLE_SECRET_REDACTION,
)
_diagnostic_filter: Final = DiagnosticProcessingFilter()
class CorrelationContextFilter(logging.Filter):
@ -633,7 +785,9 @@ class JsonFormatter(Formatter):
if record.exc_info:
json_record["stacktrace"] = record.exc_text or self.formatException(record.exc_info)
return safe_dumps(json_record, value_transform=None if _is_redacted(record) else _redact_structured_value)
return safe_dumps(
json_record if _is_redacted(record) or not _ENABLE_SECRET_REDACTION else _redact_json_record(json_record)
)
class CorrelationPlainFormatter(logging.Formatter):
@ -663,7 +817,7 @@ def _setup_json_exception_handlers(formatter):
# Create a handler with JSON formatting for exceptions
error_handler: Final = logging.StreamHandler()
error_handler.setFormatter(formatter)
error_handler.addFilter(_stdout_truncation_filter)
error_handler.addFilter(_diagnostic_filter)
error_handler.addFilter(_secret_filter)
error_handler.addFilter(_correlation_filter)
@ -734,10 +888,10 @@ verbose_logger.addHandler(handler)
# Filters attached to the logger, not the handler, survive callers swapping in their own
# handlers (JSON mode, uvicorn log config, a host app's root handler).
verbose_router_logger.addFilter(_stdout_truncation_filter)
verbose_proxy_logger.addFilter(_stdout_truncation_filter)
verbose_proxy_stdout_logger.addFilter(_stdout_truncation_filter)
verbose_logger.addFilter(_stdout_truncation_filter)
verbose_router_logger.addFilter(_diagnostic_filter)
verbose_proxy_logger.addFilter(_diagnostic_filter)
verbose_proxy_stdout_logger.addFilter(_diagnostic_filter)
verbose_logger.addFilter(_diagnostic_filter)
def _suppress_loggers():

View file

@ -2375,6 +2375,11 @@ def default_image_cost_calculator(
model_name_without_custom_llm_provider = model.replace(f"{custom_llm_provider}/", "")
base_model_name = f"{custom_llm_provider}/{size_str}/{model_name_without_custom_llm_provider}"
model_name_with_quality: Final = f"{quality}/{base_model_name}" if quality else base_model_name
provider_first_model_name_with_quality: Final = (
f"{custom_llm_provider}/{quality}/{size_str}/{model_name_without_custom_llm_provider or model}"
if quality and custom_llm_provider
else None
)
# gpt-image-1 models use low, medium, high quality. If user did not specify quality, use medium fot gpt-image-1 model family
model_name_with_v2_quality: Final = f"{ImageGenerationRequestQuality.HIGH.value}/{base_model_name}"
@ -2386,6 +2391,7 @@ def default_image_cost_calculator(
models_to_check: Final = (
model_name_with_quality,
provider_first_model_name_with_quality,
base_model_name,
model_name_with_v2_quality,
model_with_quality_without_provider,

View file

@ -10,6 +10,7 @@ import re
from typing import Final
from litellm.constants import MINIMUM_CUSTOM_KEY_LENGTH
from litellm.rust_bridge import diagnostics
REDACTED: Final = "REDACTED"
@ -87,9 +88,13 @@ def _build_secret_patterns() -> "re.Pattern[str]":
_SECRET_RE: Final = _build_secret_patterns()
def _python_redact_string(value: str) -> str:
return _SECRET_RE.sub(REDACTED, value)
def redact_string(value: str) -> str:
"""Scrub known secret/credential patterns from *value* and return the result."""
return _SECRET_RE.sub(REDACTED, value)
return diagnostics.run(lambda native: native.redact_text(value), lambda: _python_redact_string(value))
_UNIX_SYSTEM_PATH: Final = r"/(?:etc|var|opt|usr|home|root|private|Users|tmp|mnt|srv)/[^\s'\"\)\]}>,]+"
@ -105,15 +110,21 @@ _INTERNAL_DETAIL_RE: Final = re.compile(
_TRACEBACK_MARKER: Final = "Traceback (most recent call last):"
def redact_internal_details(value: str) -> str:
def _python_redact_internal_details(value: str) -> str:
"""Drop an embedded traceback and scrub filesystem paths and internal hostnames,
on top of redact_string(). For client-facing messages only: server logs keep this detail."""
marker_index: Final = value.find(_TRACEBACK_MARKER)
without_traceback: Final = value[:marker_index].rstrip() if marker_index != -1 else value
return _INTERNAL_DETAIL_RE.sub(REDACTED, redact_string(without_traceback))
return _INTERNAL_DETAIL_RE.sub(REDACTED, _python_redact_string(without_traceback))
def redact_structured_value(key: str | None, value: str) -> str:
def redact_internal_details(value: str) -> str:
return diagnostics.run(
lambda native: native.redact_client_message(value), lambda: _python_redact_internal_details(value)
)
def _python_redact_structured_value(key: str | None, value: str) -> str:
"""Scrub *value* as it appeared under *key* inside a structured record.
redact_string() replaces a whole ``key: value`` span with REDACTED, which is
@ -122,8 +133,15 @@ def redact_structured_value(key: str | None, value: str) -> str:
repr would, so the key-name patterns still fire, but collapses only the value
so the caller's structure survives.
"""
scrubbed: Final = redact_string(value)
scrubbed: Final = _python_redact_string(value)
if scrubbed != value or key is None:
return scrubbed
rendered: Final = f"'{key}': '{value}'"
return REDACTED if redact_string(rendered) != rendered else value
return REDACTED if _python_redact_string(rendered) != rendered else value
def redact_structured_value(key: str | None, value: str) -> str:
return diagnostics.run(
lambda native: native.redact_structured_text(key, value),
lambda: _python_redact_structured_value(key, value),
)

View file

@ -629,17 +629,34 @@ class AmazonConverseConfig(BaseConfig):
"""
return self._is_deepseek_r1_model(model=model, base_model=base_model)
@classmethod
def _supports_sampling_params(cls, model: str) -> bool:
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
base_model: Final = BedrockModelInfo.get_base_model(model)
if base_model.startswith("anthropic"):
return True
candidates: Final = (model, *(f"{prefix}{base_model}" for prefix in ("global.", "us.", "eu.")))
for candidate in candidates:
if (
flag := AnthropicModelInfo._get_model_capability( # pyright: ignore[reportPrivateUsage] # Shared API
candidate, "supports_sampling_params"
)
) is not None:
return flag
return True
def get_supported_openai_params(self, model: str) -> list[str]:
from litellm.utils import supports_function_calling
supports_sampling: Final = self._supports_sampling_params(model)
supported_params: Final = [
"max_tokens",
"max_completion_tokens",
"stream",
"stream_options",
"stop",
"temperature",
"top_p",
*(("temperature", "top_p") if supports_sampling else ()),
"extra_headers",
"response_format",
"requestMetadata",
@ -1019,14 +1036,26 @@ class AmazonConverseConfig(BaseConfig):
value = [value]
optional_params["stopSequences"] = value
if param == "temperature" or param == "top_p":
AnthropicConfig._apply_sampling_param(
optional_params=optional_params,
model=model,
param=param,
value=value,
drop_params=drop_params,
output_key="topP" if param == "top_p" else param,
)
if base_model.startswith("anthropic"):
AnthropicConfig._apply_sampling_param(
optional_params=optional_params,
model=model,
param=param,
value=value,
drop_params=drop_params,
output_key="topP" if param == "top_p" else param,
)
elif not self._supports_sampling_params(model):
if not (litellm.drop_params or drop_params):
raise litellm.utils.UnsupportedParamsError(
message=(
f"{model} does not support {param}={value}. "
"To drop unsupported params, set `litellm.drop_params = True`."
),
status_code=400,
)
else:
optional_params["topP" if param == "top_p" else param] = value
if param == "tools" and isinstance(value, list):
self._apply_tool_call_transformation(
tools=cast(list[OpenAIChatCompletionToolParam], value),

View file

@ -231,6 +231,12 @@ async def make_call(
sync_stream=False,
)
completion_stream = decoder.aiter_bytes(response.aiter_bytes(chunk_size=stream_chunk_size))
elif bedrock_invoke_provider == "moonshot":
decoder = AmazonOpenAICompatibleStreamDecoder(
model=model,
sync_stream=False,
)
completion_stream = decoder.aiter_bytes(response.aiter_bytes(chunk_size=stream_chunk_size))
else:
decoder = AWSEventStreamDecoder(model=model, json_mode=json_mode)
completion_stream = decoder.aiter_bytes(response.aiter_bytes(chunk_size=stream_chunk_size))
@ -329,6 +335,12 @@ def make_sync_call(
sync_stream=True,
)
completion_stream = decoder.iter_bytes(response.iter_bytes(chunk_size=stream_chunk_size))
elif bedrock_invoke_provider == "moonshot":
decoder = AmazonOpenAICompatibleStreamDecoder(
model=model,
sync_stream=True,
)
completion_stream = decoder.iter_bytes(response.iter_bytes(chunk_size=stream_chunk_size))
else:
decoder = AWSEventStreamDecoder(model=model, json_mode=json_mode)
completion_stream = decoder.iter_bytes(response.iter_bytes(chunk_size=stream_chunk_size))
@ -795,6 +807,24 @@ class AmazonDeepSeekR1StreamDecoder(AWSEventStreamDecoder):
return self.deepseek_model_response_iterator.chunk_parser(chunk=chunk_data)
class AmazonOpenAICompatibleStreamDecoder(AWSEventStreamDecoder):
def __init__(
self,
model: str,
sync_stream: bool,
) -> None:
super().__init__(model=model)
from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler
self.openai_model_response_iterator = OpenAIChatCompletionStreamingHandler(
streaming_response=None,
sync_stream=sync_stream,
)
def _chunk_parser(self, chunk_data: dict[str, object]) -> ModelResponseStream:
return self.openai_model_response_iterator.chunk_parser(chunk=chunk_data)
class MockResponseIterator: # for returning ai21 streaming responses
def __init__(self, model_response, json_mode: bool | None = False):
self.model_response = model_response

View file

@ -770,9 +770,7 @@ class AmazonAnthropicClaudeMessagesConfig(
aws_decoder: Final = AmazonAnthropicClaudeMessagesStreamDecoder(
model=model,
)
completion_stream: Final = aws_decoder.aiter_bytes(
httpx_response.aiter_bytes(chunk_size=aws_decoder.DEFAULT_CHUNK_SIZE)
)
completion_stream: Final = aws_decoder.aiter_bytes(httpx_response.aiter_bytes())
# Convert decoded Bedrock events to Server-Sent Events expected by Anthropic clients.
return self.bedrock_sse_wrapper(
completion_stream=completion_stream,
@ -919,16 +917,6 @@ class AmazonAnthropicClaudeMessagesConfig(
class AmazonAnthropicClaudeMessagesStreamDecoder(AWSEventStreamDecoder):
def __init__(
self,
model: str,
) -> None:
"""
Iterator to return Bedrock invoke response in anthropic /messages format
"""
super().__init__(model=model)
self.DEFAULT_CHUNK_SIZE = 1024
def _chunk_parser(self, chunk_data: dict) -> GChunk | ModelResponseStream | dict:
"""
Parse the chunk data into anthropic /messages format

View file

@ -317,7 +317,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
Check if the model is Gemini 3 or newer.
"""
model_name = model.split("/")[-1].lower()
if not model_name:
is_vertex_fine_tuned_model: Final = model_name.isdigit() or (
model.startswith("gemini/") and not model_name.startswith("gemini-")
)
if not model_name or is_vertex_fine_tuned_model or model_name.startswith("gemma-"):
return False
# Pre-Gemini 3 models: gemini-1.x, gemini-2.x, gemini-pro, gemini-flash, gemini-exp
if re.match(r"^gemini-(?:[12](?:\.\d+)?|exp|(?:pro|flash)(?!-(?:lite-)?latest$))(?:-|$)", model_name):

File diff suppressed because it is too large Load diff

View file

@ -1004,6 +1004,8 @@ def _stamp_deployment_attribution(
return attribution
metadata.setdefault("model_info", attribution["model_info"])
metadata.setdefault("deployment", attribution["deployment"])
if isinstance(model_group, str):
metadata.setdefault("model_group", model_group)
return attribution

View file

@ -12,6 +12,22 @@ class RustUpstreamError(Exception): ...
class ForkedAfterNativeRuntimeStarted(RuntimeError): ...
class ProcessReservedForForking(RuntimeError): ...
@final
class NativeDiagnosticProcessor:
def __new__(cls, minimum_custom_key_length: int) -> NativeDiagnosticProcessor: ...
def redact_text(self, text: str) -> str: ...
def redact_structured_text(self, key: str | None, text: str) -> str: ...
def redact_client_message(self, text: str) -> str: ...
def process_diagnostic(
self,
message: str,
exception: str | None,
stack: str | None,
leaves: Sequence[tuple[str | None, str]],
policy: tuple[bool, int, int],
) -> tuple[str, str | None, str | None, list[str], bool]: ...
def scrub_access_arguments(self, arguments: Sequence[str]) -> list[str]: ...
def ocr(
request: LiteLLMOcrRequest,
args: tuple[object, ...],
@ -335,6 +351,7 @@ def reserve_process_for_forking() -> None: ...
__all__ = [
"ForkedAfterNativeRuntimeStarted",
"HuggingFaceEncoding",
"NativeDiagnosticProcessor",
"ProcessReservedForForking",
"ResponsesWebSocketConnection",
"RustBridgeDeclined",

View file

@ -86,11 +86,25 @@ class SecretManagerRule:
return isinstance(context, SecretManagerContext) and (self.systems is None or context.system in self.systems)
Context: TypeAlias = RouteContext | CacheContext | SecretManagerContext
Rule: TypeAlias = RouteRule | CacheRule | SecretManagerRule
@dataclass(frozen=True, slots=True)
class LoggerContext:
pass
@dataclass(frozen=True, slots=True)
class LoggerRule:
rollout: Rollout
def matches(self, context: Context) -> bool:
return isinstance(context, LoggerContext)
Context: TypeAlias = RouteContext | CacheContext | SecretManagerContext | LoggerContext
Rule: TypeAlias = RouteRule | CacheRule | SecretManagerRule | LoggerRule
Rules: TypeAlias = tuple[Rule, ...]
RULES: Final[Rules] = (
LoggerRule(Rollout.RUST_OPT_IN),
RouteRule(Route.OCR, Rollout.RUST_REQUIRED, providers=frozenset({"aws_textract"})),
RouteRule(Route.OCR, Rollout.RUST_OPT_OUT),
RouteRule(Route.MESSAGES, Rollout.PYTHON_ONLY),

View file

@ -0,0 +1,59 @@
from __future__ import annotations
from collections.abc import Callable, Sequence
from functools import lru_cache
from typing import Final, Protocol, TypeVar, cast
from litellm.rust_bridge.bindings import NativeBinding
ResultT: Final = TypeVar("ResultT")
class NativeDiagnosticProcessor(Protocol):
def redact_text(self, text: str) -> str: ...
def redact_structured_text(self, key: str | None, text: str) -> str: ...
def redact_client_message(self, text: str) -> str: ...
def process_diagnostic(
self,
message: str,
exception: str | None,
stack: str | None,
leaves: tuple[tuple[str | None, str], ...],
policy: tuple[bool, int, int],
) -> tuple[str, str | None, str | None, Sequence[str], bool]: ...
def scrub_access_arguments(self, arguments: tuple[str, ...]) -> Sequence[str]: ...
class NativeDiagnosticFactory(Protocol):
def __call__(self, minimum_custom_key_length: int) -> NativeDiagnosticProcessor: ...
def _as_factory(value: object) -> NativeDiagnosticFactory | None:
if not isinstance(value, type):
return None
return cast(NativeDiagnosticFactory, value) # cast-ok: PyO3 factory must be a type
PROCESSOR: Final = NativeBinding("NativeDiagnosticProcessor", validate=_as_factory)
@lru_cache(maxsize=4)
def _construct(factory: NativeDiagnosticFactory, minimum_custom_key_length: int) -> NativeDiagnosticProcessor:
return factory(minimum_custom_key_length)
def run(native: Callable[[NativeDiagnosticProcessor], ResultT], python: Callable[[], ResultT]) -> ResultT:
from litellm.constants import MINIMUM_CUSTOM_KEY_LENGTH
from litellm.rust_bridge.catalog import LoggerContext, decision
from litellm.rust_bridge.configuration import Decision
selected: Final = decision(LoggerContext())
if selected is Decision.PYTHON:
return python()
factory: Final = PROCESSOR.load()
if factory is None:
return python()
try:
return native(_construct(factory, MINIMUM_CUSTOM_KEY_LENGTH))
except Exception:
return python()

View file

@ -0,0 +1,63 @@
from __future__ import annotations
from collections.abc import Mapping
from typing import Final
from pydantic import JsonValue
from litellm._logging import (
CorrelationContextFilter,
DiagnosticProcessingFilter,
session_id_var,
set_session_id,
set_trace_id,
trace_id_var,
verbose_logger,
)
_REDACTION: Final = DiagnosticProcessingFilter()
_CORRELATION: Final = CorrelationContextFilter()
def context() -> tuple[str, str]:
return session_id_var.get(), trace_id_var.get()
def enabled(level: int) -> bool:
return verbose_logger.isEnabledFor(level)
def emit(
level: int,
message: str,
pathname: str,
lineno: int,
target: str,
fields: Mapping[str, JsonValue],
correlation: tuple[str, str],
) -> None:
if not enabled(level):
return
session_token: Final = set_session_id(correlation[0])
trace_token: Final = set_trace_id(correlation[1])
try:
record: Final = verbose_logger.makeRecord(
verbose_logger.name,
level,
pathname,
lineno,
message,
(),
None,
func=target,
extra={
"rust_target": target,
"rust_fields": dict(fields),
}, # mutable-ok: LogRecord requires JSON dict extras
)
_REDACTION.filter(record)
_CORRELATION.filter(record)
verbose_logger.handle(record)
finally:
trace_id_var.reset(trace_token)
session_id_var.reset(session_token)

View file

@ -58,12 +58,6 @@ class SecretManagerBinding:
settings_object: object
def warn(message: str) -> None:
from litellm._logging import verbose_logger
verbose_logger.warning("%s", message)
def secret_manager() -> SecretManager:
from litellm.secret_managers.main import (
_should_read_secret_from_secret_manager, # pyright: ignore[reportPrivateUsage] # canonical resolver is private

View file

@ -1,5 +1,5 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:e624c5d5e42382ce7165ddafcbbf8e6769a24cbd02ea6114b880b05ae5ba2a8d
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin

File diff suppressed because it is too large Load diff

View file

@ -1,6 +1,6 @@
[project]
name = "litellm"
version = "1.103.0"
version = "1.104.0"
description = "Library to easily interface with LLM API providers"
readme = "README.md"
requires-python = ">=3.10, <3.15"
@ -75,8 +75,8 @@ proxy = [
"mcp>=2.2.0,<3",
"httpx2>=2.5.0,<3",
"pydantic>=2.12.0,<3",
"litellm-proxy-extras==0.4.100",
"litellm-enterprise==0.1.69",
"litellm-proxy-extras==0.4.101",
"litellm-enterprise==0.1.70",
"RestrictedPython>=8.5,<9.0",
"rich>=13.9.4,<14.0",
"InquirerPy>=0.3.4,<1.0",
@ -355,7 +355,7 @@ litellm-enterprise = { workspace = true }
members = ["enterprise", "litellm-proxy-extras"]
[tool.commitizen]
version = "1.103.0"
version = "1.104.0"
version_files = [
"pyproject.toml:^version",
]

View file

@ -3,62 +3,140 @@ import { describe, expect, test } from "bun:test";
import type { Comment, GitHubApi } from "./auto-close-duplicates";
import {
FIXED_MARKER,
OPEN_PULL_REQUESTS_QUERY,
SUPERSEDED_MARKER,
closeVerdict,
closerOf,
commentFixedIssue,
describeIssue,
describeSweep,
fixOf,
fixedBody,
handleFixedIssue,
nextMinor,
parseVersion,
placement,
readConfig,
releaseCandidate,
supersededBody,
sweep,
type ClosedIssue,
type FixedConfig,
type IssueClosure,
type LinkedIssue,
type LinkedPullRequest,
type PullRequestsPage,
} from "./comment-fixed-issue";
const MERGE_COMMIT = "68c4c82ac977b48b2b81ee8d633d5771307c6162";
const ISSUE = 41750;
const mergedPr = {
__typename: "PullRequest" as const,
number: 41767,
merged: true,
baseRefName: "main",
repository: { nameWithOwner: "BerriAI/litellm" },
mergeCommit: { oid: MERGE_COMMIT },
};
type Closer = ClosedIssue["timelineItems"]["nodes"][number]["closer"];
const commitCloser = { __typename: "Commit" as const, oid: MERGE_COMMIT, repository: { nameWithOwner: "BerriAI/litellm" } };
const closedBy = (closer: Closer, state: ClosedIssue["state"] = "CLOSED"): ClosedIssue => ({
type Closer = IssueClosure["timelineItems"]["nodes"][number]["closer"];
const closure = (closer: Closer, state: IssueClosure["state"] = "CLOSED"): IssueClosure => ({
state,
timelineItems: { nodes: [{ closer }] },
});
const page = (pages: readonly (readonly LinkedPullRequest[])[], index: number): PullRequestsPage => ({
pageInfo: { hasNextPage: index + 1 < pages.length, endCursor: String(index + 1) },
nodes: pages[index] ?? [],
});
const cursorIndex = (after: string | null): number => (after === null ? 0 : Number(after));
const closedBy = (closer: Closer, state?: IssueClosure["state"], linked: readonly LinkedPullRequest[] = []): ClosedIssue => ({
...closure(closer, state),
closedByPullRequestsReferences: page([linked], 0),
});
const reopenedAt = (createdAt: string): LinkedPullRequest["reopens"] => ({ nodes: [{ createdAt }] });
const linkedIssue = (number: number, closer: Closer = mergedPr, state?: IssueClosure["state"]): LinkedIssue => ({
number,
repository: { nameWithOwner: "BerriAI/litellm" },
...closure(closer, state),
});
const links = (...issues: readonly LinkedIssue[]): LinkedPullRequest["closingIssuesReferences"] => ({ totalCount: issues.length, nodes: issues });
const openPr = (number: number, overrides: Partial<LinkedPullRequest> = {}): LinkedPullRequest => ({
number,
state: "OPEN",
baseRefName: "main",
repository: { nameWithOwner: "BerriAI/litellm" },
closingIssuesReferences: links(linkedIssue(ISSUE)),
reopens: { nodes: [] },
...overrides,
});
const pyproject = (version: string): string =>
`[project]\nname = "litellm"\nversion = "${version}"\n\n[tool.commitizen]\nversion = "${version}"\n`;
const config: FixedConfig = { repo: "BerriAI/litellm", issueNumber: 41750, defaultBranch: "main", dryRun: false };
const config: FixedConfig = { repo: "BerriAI/litellm", defaultBranch: "main", commentDryRun: false, closeDryRun: false };
const prFix = (issue = ISSUE, number = 41767) => ({ issue, source: { kind: "pull_request" as const, number, oid: MERGE_COMMIT } });
const commitFix = (issue = ISSUE) => ({ issue, source: { kind: "commit" as const, oid: MERGE_COMMIT } });
const oneFixBody = supersededBody([prFix()], "main");
const noPause = async (): Promise<void> => {};
const supersededComment: Comment = {
id: 2,
body: `${SUPERSEDED_MARKER}\n#41750 was fixed by #41767 on main, so this pull request is closed. Reopen it if something was missed.`,
created_at: "2026-09-18T00:00:00Z",
user: { type: "Bot", login: "github-actions[bot]" },
};
interface World {
readonly issue?: ClosedIssue | null;
readonly comments?: readonly Comment[];
readonly pullRequestComments?: Readonly<Record<number, readonly Comment[]>>;
readonly openPullRequests?: readonly (readonly LinkedPullRequest[])[];
readonly linkedPullRequests?: readonly (readonly LinkedPullRequest[])[];
readonly version?: string;
// Which existing rc.1 tags contain the merge commit; a tag absent from the map does not exist
readonly tags?: Readonly<Record<string, boolean>>;
readonly reachable?: Readonly<Record<string, readonly string[]>>;
}
function fakeApi(world: World = {}): { readonly api: GitHubApi; readonly writes: string[] } {
const writes: string[] = [];
const tags = world.tags ?? {};
const reachable = world.reachable ?? { main: [MERGE_COMMIT] };
const pages = world.openPullRequests ?? [];
const api: GitHubApi = {
request: async <T>(method: string, path: string, body?: object): Promise<T> => {
if (method === "POST" && path === "/graphql") {
return { data: { repository: { issue: world.issue === undefined ? closedBy(mergedPr) : world.issue } } } as T;
const { query, variables } = body as { query: string; variables: { after: string | null } };
if (query === OPEN_PULL_REQUESTS_QUERY) {
return { data: { repository: { pullRequests: page(pages, cursorIndex(variables.after)) } } } as T;
}
const issue = world.issue === undefined ? closedBy(mergedPr) : world.issue;
if (issue === null || world.linkedPullRequests === undefined) {
return { data: { repository: { issue } } } as T;
}
const closedByPullRequestsReferences = page(world.linkedPullRequests, cursorIndex(variables.after));
return { data: { repository: { issue: { ...issue, closedByPullRequestsReferences } } } } as T;
}
if (method !== "GET") {
writes.push(`${method} ${path} ${JSON.stringify(body)}`);
return {} as T;
}
if (path.startsWith("/repos/BerriAI/litellm/issues/41750/comments")) {
return (world.comments ?? []) as T;
const comments = /^\/repos\/BerriAI\/litellm\/issues\/(\d+)\/comments/.exec(path);
if (comments !== null) {
const number = Number(comments[1]);
return ((number === ISSUE ? world.comments : world.pullRequestComments?.[number]) ?? []) as T;
}
if (path === `/repos/BerriAI/litellm/contents/pyproject.toml?ref=${MERGE_COMMIT}`) {
return { content: btoa(pyproject(world.version ?? "1.103.0")).replace(/(.{60})/g, "$1\n") } as T;
@ -68,6 +146,10 @@ function fakeApi(world: World = {}): { readonly api: GitHubApi; readonly writes:
return (matching[1] in tags ? [{ ref: `refs/tags/${matching[1]}` }] : []) as T;
}
const compare = /^\/repos\/BerriAI\/litellm\/compare\/(.+)\.\.\.(.+)$/.exec(path);
const branch = compare === null ? undefined : reachable[compare[1] ?? ""];
if (compare !== null && branch !== undefined) {
return { status: branch.includes(compare[2] ?? "") ? "behind" : "diverged" } as T;
}
if (compare !== null && compare[2] === MERGE_COMMIT) {
return { status: tags[compare[1]] ? "behind" : "ahead" } as T;
}
@ -79,23 +161,28 @@ function fakeApi(world: World = {}): { readonly api: GitHubApi; readonly writes:
describe("closerOf", () => {
test("a pull request merged into the default branch is the fix", () => {
expect(closerOf(closedBy(mergedPr), "main")).toEqual({ kind: "pull_request", number: 41767, mergeCommit: MERGE_COMMIT });
expect(closerOf(closedBy(mergedPr), "BerriAI/litellm", "main")).toEqual({ kind: "pull_request", number: 41767, mergeCommit: MERGE_COMMIT });
});
test("an issue closed by hand, by a commit, or by an unmerged pull request gets no comment", () => {
expect(closerOf(closedBy(null), "main")).toEqual({ kind: "skip", reason: "closed by hand, not by a pull request" });
expect(closerOf(closedBy({ __typename: "Commit", oid: MERGE_COMMIT }), "main").kind).toBe("skip");
expect(closerOf(closedBy({ ...mergedPr, merged: false }), "main").kind).toBe("skip");
expect(closerOf(closedBy({ ...mergedPr, mergeCommit: null }), "main").kind).toBe("skip");
expect(closerOf(closedBy(null), "BerriAI/litellm", "main")).toEqual({ kind: "skip", reason: "closed by hand, not by a pull request" });
expect(closerOf(closedBy(commitCloser), "BerriAI/litellm", "main").kind).toBe("skip");
expect(closerOf(closedBy({ ...mergedPr, merged: false }), "BerriAI/litellm", "main").kind).toBe("skip");
expect(closerOf(closedBy({ ...mergedPr, mergeCommit: null }), "BerriAI/litellm", "main").kind).toBe("skip");
});
test("a pull request merged into a release branch is not a fix on main", () => {
const verdict = closerOf(closedBy({ ...mergedPr, baseRefName: "release/1.102.0rc2" }), "main");
const verdict = closerOf(closedBy({ ...mergedPr, baseRefName: "release/1.102.0rc2" }), "BerriAI/litellm", "main");
expect(verdict).toEqual({ kind: "skip", reason: "#41767 merged into release/1.102.0rc2, not main" });
});
test("an issue reopened after the close event is left alone", () => {
expect(closerOf(closedBy(mergedPr, "OPEN"), "main")).toEqual({ kind: "skip", reason: "the issue is open again" });
expect(closerOf(closedBy(mergedPr, "OPEN"), "BerriAI/litellm", "main")).toEqual({ kind: "skip", reason: "the issue is open again" });
});
test("a pull request merged in a fork closes the issue on GitHub but is no fix here", () => {
const forkPr = { ...mergedPr, number: 9, repository: { nameWithOwner: "someone/litellm" } };
expect(closerOf(closedBy(forkPr), "BerriAI/litellm", "main")).toEqual({ kind: "skip", reason: "closed by someone/litellm#9, a pull request in another repository" });
});
});
@ -173,44 +260,327 @@ describe("fixedBody", () => {
});
});
describe("commentFixedIssue", () => {
describe("fixOf", () => {
test("a merged pull request or a commit is a fix, whatever branch it was merged into", () => {
expect(fixOf(closure(mergedPr), "BerriAI/litellm")).toEqual({ kind: "pull_request", number: 41767, oid: MERGE_COMMIT });
expect(fixOf(closure({ ...mergedPr, baseRefName: "litellm_internal_staging" }), "BerriAI/litellm")).toEqual({ kind: "pull_request", number: 41767, oid: MERGE_COMMIT });
expect(fixOf(closure(commitCloser), "BerriAI/litellm")).toEqual({ kind: "commit", oid: MERGE_COMMIT });
});
test("a hand close, a fork's pull request, an unmerged pull request, or a reopened issue is no fix", () => {
expect(fixOf(closure(null), "BerriAI/litellm")).toEqual({ kind: "skip", reason: "was closed by hand" });
expect(fixOf(closure({ ...mergedPr, repository: { nameWithOwner: "someone/litellm" } }), "BerriAI/litellm")).toEqual({ kind: "skip", reason: "was closed from someone/litellm" });
expect(fixOf(closure({ ...commitCloser, repository: { nameWithOwner: "someone/litellm" } }), "BerriAI/litellm")).toEqual({ kind: "skip", reason: "was closed from someone/litellm" });
expect(fixOf(closure({ ...mergedPr, merged: false }), "BerriAI/litellm")).toEqual({ kind: "skip", reason: "was closed by #41767, which is not merged" });
expect(fixOf(closure({ ...mergedPr, mergeCommit: null }), "BerriAI/litellm").kind).toBe("skip");
expect(fixOf(closure(mergedPr, "OPEN"), "BerriAI/litellm")).toEqual({ kind: "skip", reason: "is open again" });
});
});
describe("closeVerdict", () => {
test("an open pull request whose only linked issue was fixed by a merged pull request is a candidate", () => {
expect(closeVerdict(openPr(41760), config)).toEqual({ kind: "candidate", fixes: [prFix()] });
});
test("every linked issue has to be fixed, and each fix is named", () => {
const both = openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE), linkedIssue(41751, commitCloser)) });
expect(closeVerdict(both, config)).toEqual({ kind: "candidate", fixes: [prFix(), commitFix(41751)] });
});
test("a pull request that still links an open issue keeps its work", () => {
const stillOpen = openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE), linkedIssue(41751, null, "OPEN")) });
expect(closeVerdict(stillOpen, config)).toEqual({ kind: "skip", reason: "still linked to open #41751" });
});
test("a linked issue closed by hand or by an unmerged pull request is not a fix that supersedes the pull request", () => {
expect(closeVerdict(openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE, null)) }), config)).toEqual({
kind: "skip",
reason: `#${ISSUE} was closed by hand`,
});
const unmerged = openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE, { ...mergedPr, merged: false })) });
expect(closeVerdict(unmerged, config)).toEqual({ kind: "skip", reason: `#${ISSUE} was closed by #41767, which is not merged` });
});
test("a pull request linking an issue in another repository, or more issues than the query reads, is left alone", () => {
const foreign = { ...linkedIssue(41751), repository: { nameWithOwner: "mlflow/mlflow" } };
expect(closeVerdict(openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE), foreign) }), config)).toEqual({
kind: "skip",
reason: "links mlflow/mlflow#41751",
});
const truncated = openPr(41760, { closingIssuesReferences: { totalCount: 11, nodes: [linkedIssue(ISSUE)] } });
expect(closeVerdict(truncated, config)).toEqual({ kind: "skip", reason: "links 11 issues, more than the 10 this workflow reads" });
});
test("a pull request against a release line is a backport and stays open, one against a retired development branch does not", () => {
for (const base of ["release/v1.102.0-rc.2", "stable/v1.83.14", "v_1_83_3_stable_patch"]) {
expect(closeVerdict(openPr(41760, { baseRefName: base }), config)).toEqual({ kind: "skip", reason: `targets the release line ${base}` });
}
expect(closeVerdict(openPr(41760, { baseRefName: "litellm_internal_staging" }), config).kind).toBe("candidate");
});
test("a pull request in a fork, one that is not open, or one linking no issue is left alone", () => {
expect(closeVerdict(openPr(1, { repository: { nameWithOwner: "someone/litellm" } }), config)).toEqual({ kind: "skip", reason: "lives in someone/litellm" });
expect(closeVerdict(openPr(41767, { state: "MERGED" }), config)).toEqual({ kind: "skip", reason: "is merged" });
expect(closeVerdict(openPr(41760, { closingIssuesReferences: links() }), config)).toEqual({ kind: "skip", reason: "links no issue" });
});
});
describe("supersededBody", () => {
test("names each issue with the pull request or commit that fixed it, invites a reopen, and carries the marker the rerun looks for", () => {
expect(oneFixBody.startsWith(SUPERSEDED_MARKER)).toBe(true);
expect(oneFixBody).toContain("#41750 was fixed by #41767 on main");
expect(oneFixBody).toContain("Reopen it");
expect(supersededBody([prFix(), commitFix(41751)], "main")).toContain("#41750 was fixed by #41767 and #41751 by commit 68c4c82ac9 on main");
});
test("stays within the 25-word comment rule for one and two fixes", () => {
for (const fixes of [[prFix()], [commitFix()], [prFix(), commitFix(41751)]]) {
const words = supersededBody(fixes, "main").replace(SUPERSEDED_MARKER, "").trim().split(/\s+/);
expect(words.length).toBeGreaterThanOrEqual(15);
expect(words.length).toBeLessThanOrEqual(25);
}
});
});
describe("handleFixedIssue", () => {
test("a real run posts one comment naming the pull request and the release", async () => {
const { api, writes } = fakeApi();
const verdict = await commentFixedIssue(api, config);
expect(verdict).toMatchObject({ kind: "commented", pullRequest: 41767, tag: "v1.103.0-rc.1" });
const { comment, pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(comment).toMatchObject({ kind: "commented", pullRequest: 41767, tag: "v1.103.0-rc.1" });
expect(pullRequests).toEqual([]);
expect(writes).toHaveLength(1);
expect(writes[0]).toContain("POST /repos/BerriAI/litellm/issues/41750/comments");
expect(writes[0]).toContain("Fixed by #41767. This ships in v1.103.0-rc.1 and up");
});
test("a dry run renders the comment and writes nothing", async () => {
const { api, writes } = fakeApi();
const verdict = await commentFixedIssue(api, { ...config, dryRun: true });
expect(verdict.kind).toBe("commented");
expect(writes).toEqual([]);
test("every other open pull request linked to the fixed issue is commented on and then closed, a pause before every write", async () => {
const { api, writes } = fakeApi({ issue: closedBy(mergedPr, "CLOSED", [openPr(41760), openPr(41761)]) });
let pauses = 0;
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, async () => {
pauses += 1;
});
expect(pullRequests.map((pullRequest) => pullRequest.kind)).toEqual(["closed", "closed"]);
expect(writes).toEqual([
expect.stringContaining("POST /repos/BerriAI/litellm/issues/41750/comments"),
`POST /repos/BerriAI/litellm/issues/41760/comments ${JSON.stringify({ body: oneFixBody })}`,
'PATCH /repos/BerriAI/litellm/pulls/41760 {"state":"closed"}',
expect.stringContaining("POST /repos/BerriAI/litellm/issues/41761/comments"),
'PATCH /repos/BerriAI/litellm/pulls/41761 {"state":"closed"}',
]);
expect(pauses).toBe(4);
});
test("an issue that already carries the comment is not commented twice", async () => {
test("the fixed-in comment and the closing are gated separately: a close dry run still posts the comment and closes nothing", async () => {
const { api, writes } = fakeApi({ issue: closedBy(mergedPr, "CLOSED", [openPr(41760)]) });
let pauses = 0;
const outcome = await handleFixedIssue(api, { ...config, closeDryRun: true }, ISSUE, async () => {
pauses += 1;
});
expect(outcome.comment.kind).toBe("commented");
expect(outcome.pullRequests).toEqual([{ kind: "closed", number: 41760, body: oneFixBody }]);
expect(writes).toHaveLength(1);
expect(writes[0]).toContain("/issues/41750/comments");
expect(pauses).toBe(0);
});
test("a comment dry run still closes the pull requests for real", async () => {
const { api, writes } = fakeApi({ issue: closedBy(mergedPr, "CLOSED", [openPr(41760)]) });
const outcome = await handleFixedIssue(api, { ...config, commentDryRun: true }, ISSUE, noPause);
expect(outcome.comment.kind).toBe("commented");
expect(writes).toEqual([expect.stringContaining("/issues/41760/comments"), 'PATCH /repos/BerriAI/litellm/pulls/41760 {"state":"closed"}']);
});
test("an issue that already carries the fixed-in comment is not commented twice, and its linked pull requests still get closed", async () => {
const existing: Comment = {
id: 1,
body: `${FIXED_MARKER}\nFixed by #41767. This ships in v1.103.0-rc.1 and up.`,
created_at: "2026-09-18T00:00:00Z",
user: { type: "Bot", login: "github-actions[bot]" },
};
const { api, writes } = fakeApi({ comments: [existing] });
expect(await commentFixedIssue(api, config)).toEqual({ kind: "skip", reason: "already carries a fixed-in comment" });
const { api, writes } = fakeApi({ comments: [existing], issue: closedBy(mergedPr, "CLOSED", [openPr(41760)]) });
const outcome = await handleFixedIssue(api, config, ISSUE, noPause);
expect(outcome.comment).toEqual({ kind: "skip", reason: "already carries a fixed-in comment" });
expect(writes).toEqual([expect.stringContaining("/issues/41760/comments"), 'PATCH /repos/BerriAI/litellm/pulls/41760 {"state":"closed"}']);
});
test("a fix merged into the retired development branch or pushed as a commit counts once it is on the default branch", async () => {
const stagingPr = { ...mergedPr, baseRefName: "litellm_internal_staging" };
const staging = openPr(41760, { baseRefName: "litellm_internal_staging", closingIssuesReferences: links(linkedIssue(ISSUE, stagingPr)) });
const byCommit = openPr(41761, { closingIssuesReferences: links(linkedIssue(41751, commitCloser)) });
const { api, writes } = fakeApi({ issue: closedBy(mergedPr, "CLOSED", [staging, byCommit]) });
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(pullRequests).toEqual([
{ kind: "closed", number: 41760, body: oneFixBody },
{ kind: "closed", number: 41761, body: supersededBody([commitFix(41751)], "main") },
]);
expect(writes).toHaveLength(5);
expect(writes[3]).toContain("#41751 was fixed by commit 68c4c82ac9 on main");
});
test("a fix whose commit never reached the default branch supersedes nothing", async () => {
const { api, writes } = fakeApi({ issue: closedBy(mergedPr, "CLOSED", [openPr(41760)]), reachable: { main: [] } });
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(pullRequests).toEqual([{ kind: "skip", number: 41760, reason: "#41750 was fixed by #41767, which is not on main" }]);
expect(writes).toHaveLength(1);
expect(writes[0]).toContain("/issues/41750/comments");
});
test("the default branch comes from the config for the containment check and the comment alike", async () => {
const stagingConfig = { ...config, defaultBranch: "litellm_internal_staging" };
const closer = { ...mergedPr, baseRefName: "litellm_internal_staging" };
const { api, writes } = fakeApi({
issue: closedBy(closer, "CLOSED", [openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE, closer)) })]),
reachable: { litellm_internal_staging: [MERGE_COMMIT] },
});
const { pullRequests } = await handleFixedIssue(api, stagingConfig, ISSUE, noPause);
expect(pullRequests).toEqual([{ kind: "closed", number: 41760, body: supersededBody([prFix()], "litellm_internal_staging") }]);
expect(writes[1]).toContain("on litellm_internal_staging, so this pull request is closed");
});
test("the closer sits in the linked list as merged and gets neither a line nor a write", async () => {
const { api, writes } = fakeApi({ issue: closedBy(mergedPr, "CLOSED", [openPr(41767, { state: "MERGED" }), openPr(41760)]) });
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(pullRequests).toEqual([{ kind: "closed", number: 41760, body: oneFixBody }]);
expect(writes.map((write) => write.split(" ")[1])).toEqual([
"/repos/BerriAI/litellm/issues/41750/comments",
"/repos/BerriAI/litellm/issues/41760/comments",
"/repos/BerriAI/litellm/pulls/41760",
]);
});
test("a pull request this workflow closed once and its author reopened stays open", async () => {
const reopened = openPr(41760, { reopens: reopenedAt("2026-09-19T00:00:00Z") });
const { api, writes } = fakeApi({
issue: closedBy(mergedPr, "CLOSED", [reopened, openPr(41761)]),
pullRequestComments: { 41760: [supersededComment] },
});
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(pullRequests[0]).toEqual({ kind: "skip", number: 41760, reason: "was closed by this workflow once and reopened" });
expect(pullRequests[1]?.kind).toBe("closed");
expect(writes.filter((write) => write.includes("41760"))).toEqual([]);
});
test("a pull request whose comment landed but whose close failed is closed on the next run without a second comment", async () => {
const reopenedBeforeTheComment = openPr(41761, { reopens: reopenedAt("2026-09-17T00:00:00Z") });
const { api, writes } = fakeApi({
issue: closedBy(mergedPr, "CLOSED", [openPr(41760), reopenedBeforeTheComment]),
pullRequestComments: { 41760: [supersededComment], 41761: [supersededComment] },
});
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(pullRequests).toEqual([
{ kind: "closed", number: 41760, body: supersededComment.body },
{ kind: "closed", number: 41761, body: supersededComment.body },
]);
expect(writes.filter((write) => write.includes("/4176"))).toEqual([
'PATCH /repos/BerriAI/litellm/pulls/41760 {"state":"closed"}',
'PATCH /repos/BerriAI/litellm/pulls/41761 {"state":"closed"}',
]);
});
test("a superseded marker pasted by anyone but the workflow neither keeps a pull request open nor replaces its comment", async () => {
const forged: Comment = { ...supersededComment, id: 3, user: { type: "User", login: "someone" } };
const { api, writes } = fakeApi({
issue: closedBy(mergedPr, "CLOSED", [openPr(41760, { reopens: reopenedAt("2026-09-19T00:00:00Z") })]),
pullRequestComments: { 41760: [forged] },
});
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(pullRequests).toEqual([{ kind: "closed", number: 41760, body: oneFixBody }]);
expect(writes.filter((write) => write.includes("/41760"))).toEqual([
`POST /repos/BerriAI/litellm/issues/41760/comments ${JSON.stringify({ body: oneFixBody })}`,
'PATCH /repos/BerriAI/litellm/pulls/41760 {"state":"closed"}',
]);
});
test("every page of linked pull requests is read, not just the first", async () => {
const { api } = fakeApi({ linkedPullRequests: [[openPr(41760)], [openPr(41761)], [openPr(41762)]] });
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(pullRequests).toEqual([
{ kind: "closed", number: 41760, body: oneFixBody },
{ kind: "closed", number: 41761, body: oneFixBody },
{ kind: "closed", number: 41762, body: oneFixBody },
]);
});
test("a linked pull request from a fork or one still tied to another open issue is reported, not closed", async () => {
const fork = openPr(1, { repository: { nameWithOwner: "someone/litellm" } });
const busy = openPr(41762, { closingIssuesReferences: links(linkedIssue(ISSUE), linkedIssue(41751, null, "OPEN")) });
const { api, writes } = fakeApi({ issue: closedBy(mergedPr, "CLOSED", [fork, busy]) });
const { pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(pullRequests).toEqual([
{ kind: "skip", number: 1, reason: "lives in someone/litellm" },
{ kind: "skip", number: 41762, reason: "still linked to open #41751" },
]);
expect(writes).toHaveLength(1);
});
test("a hand-closed issue gets no comment and leaves its linked pull requests open with the reason on each", async () => {
const byHand = openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE, null)) });
const { api, writes } = fakeApi({ issue: closedBy(null, "CLOSED", [byHand]) });
const outcome = await handleFixedIssue(api, config, ISSUE, noPause);
expect(outcome.comment).toEqual({ kind: "skip", reason: "closed by hand, not by a pull request" });
expect(outcome.pullRequests).toEqual([{ kind: "skip", number: 41760, reason: `#${ISSUE} was closed by hand` }]);
expect(writes).toEqual([]);
});
test("a hand-closed issue never reaches the release lookup or the API writes", async () => {
const { api, writes } = fakeApi({ issue: closedBy(null) });
expect((await commentFixedIssue(api, config)).kind).toBe("skip");
test("an issue closed by a commit on the default branch gets no comment but still closes its linked pull requests", async () => {
const byCommit = openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE, commitCloser)) });
const { api, writes } = fakeApi({ issue: closedBy(commitCloser, "CLOSED", [byCommit]) });
const { comment, pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(comment).toEqual({ kind: "skip", reason: "closed by commit 68c4c82ac9, not by a pull request" });
expect(pullRequests).toEqual([{ kind: "closed", number: 41760, body: supersededBody([commitFix()], "main") }]);
expect(writes).toEqual([expect.stringContaining("/issues/41760/comments"), 'PATCH /repos/BerriAI/litellm/pulls/41760 {"state":"closed"}']);
expect(writes[0]).toContain("#41750 was fixed by commit 68c4c82ac9 on main");
});
test("an issue closed from the retired development branch gets no comment but still closes its linked pull requests once the fix is on the default branch", async () => {
const stagingPr = { ...mergedPr, baseRefName: "litellm_internal_staging" };
const staging = openPr(41760, { closingIssuesReferences: links(linkedIssue(ISSUE, stagingPr)) });
const { api, writes } = fakeApi({ issue: closedBy(stagingPr, "CLOSED", [staging]) });
const { comment, pullRequests } = await handleFixedIssue(api, config, ISSUE, noPause);
expect(comment).toEqual({ kind: "skip", reason: "#41767 merged into litellm_internal_staging, not main" });
expect(pullRequests).toEqual([{ kind: "closed", number: 41760, body: oneFixBody }]);
expect(writes.map((write) => write.split(" ")[1])).toEqual(["/repos/BerriAI/litellm/issues/41760/comments", "/repos/BerriAI/litellm/pulls/41760"]);
});
test("an issue that is open again gets no comment and its linked pull requests are neither read nor touched", async () => {
const { api, writes } = fakeApi({ issue: closedBy(mergedPr, "OPEN", [openPr(41760)]) });
const outcome = await handleFixedIssue(api, config, ISSUE, noPause);
expect(outcome).toEqual({ comment: { kind: "skip", reason: "the issue is open again" }, pullRequests: [] });
expect(writes).toEqual([]);
});
test("a number that is not an issue in the repository is a skip", async () => {
const { api, writes } = fakeApi({ issue: null });
expect(await commentFixedIssue(api, config)).toEqual({ kind: "skip", reason: "not an issue in this repository" });
const outcome = await handleFixedIssue(api, config, ISSUE, noPause);
expect(outcome.comment).toEqual({ kind: "skip", reason: "not an issue in this repository" });
expect(writes).toEqual([]);
});
});
describe("sweep", () => {
test("walks every page of open pull requests and closes the ones whose linked issues were all fixed", async () => {
const unlinked = openPr(41700, { closingIssuesReferences: links() });
const busy = openPr(41701, { closingIssuesReferences: links(linkedIssue(41751, null, "OPEN")) });
const { api, writes } = fakeApi({ openPullRequests: [[unlinked, openPr(41760)], [busy, openPr(41761)]] });
const outcome = await sweep(api, { ...config, commentDryRun: true }, noPause);
expect(outcome.considered).toBe(4);
expect(outcome.pullRequests).toEqual([
{ kind: "closed", number: 41760, body: oneFixBody },
{ kind: "skip", number: 41701, reason: "still linked to open #41751" },
{ kind: "closed", number: 41761, body: oneFixBody },
]);
expect(writes).toEqual([
expect.stringContaining("POST /repos/BerriAI/litellm/issues/41760/comments"),
'PATCH /repos/BerriAI/litellm/pulls/41760 {"state":"closed"}',
expect.stringContaining("POST /repos/BerriAI/litellm/issues/41761/comments"),
'PATCH /repos/BerriAI/litellm/pulls/41761 {"state":"closed"}',
]);
});
test("a sweep dry run lists what it would close and writes nothing", async () => {
const { api, writes } = fakeApi({ openPullRequests: [[openPr(41760)]] });
const outcome = await sweep(api, { ...config, closeDryRun: true }, noPause);
expect(outcome.pullRequests.map((pullRequest) => pullRequest.kind)).toEqual(["closed"]);
expect(writes).toEqual([]);
});
});
@ -218,10 +588,24 @@ describe("commentFixedIssue", () => {
describe("readConfig", () => {
const env = { GITHUB_TOKEN: "t", GITHUB_REPOSITORY: "BerriAI/litellm", ISSUE_NUMBER: "41750", DEFAULT_BRANCH: "main" };
test("reads the four inputs and treats anything but the literal true as a real run", () => {
expect(readConfig(env)).toEqual({ token: "t", repo: "BerriAI/litellm", issueNumber: 41750, defaultBranch: "main", dryRun: false });
expect(readConfig({ ...env, DRY_RUN: "true" }).dryRun).toBe(true);
expect(readConfig({ ...env, DRY_RUN: "false" }).dryRun).toBe(false);
test("reads the inputs and treats anything but the literal true as a real run for each gate", () => {
expect(readConfig(env)).toEqual({
token: "t",
repo: "BerriAI/litellm",
defaultBranch: "main",
commentDryRun: false,
closeDryRun: false,
run: { kind: "issue", number: 41750 },
});
expect(readConfig({ ...env, DRY_RUN: "true" })).toMatchObject({ commentDryRun: true, closeDryRun: false });
expect(readConfig({ ...env, CLOSE_PRS_DRY_RUN: "true" })).toMatchObject({ commentDryRun: false, closeDryRun: true });
expect(readConfig({ ...env, DRY_RUN: "false", CLOSE_PRS_DRY_RUN: "false" })).toMatchObject({ commentDryRun: false, closeDryRun: false });
});
test("a sweep needs no issue number and anything but the literal true is an issue run", () => {
expect(readConfig({ ...env, ISSUE_NUMBER: undefined, SWEEP: "true" }).run).toEqual({ kind: "sweep" });
expect(readConfig({ ...env, SWEEP: "false" }).run).toEqual({ kind: "issue", number: 41750 });
expect(() => readConfig({ ...env, ISSUE_NUMBER: "", SWEEP: "false" })).toThrow("dispatch with an issue_number or with sweep ticked");
});
test("refuses a missing token, repo, branch or a bad issue number", () => {
@ -232,3 +616,31 @@ describe("readConfig", () => {
expect(() => readConfig({ ...env, ISSUE_NUMBER: "abc" })).toThrow("ISSUE_NUMBER");
});
});
describe("step summary", () => {
const closed = { kind: "closed" as const, number: 41760, body: oneFixBody };
const left = { kind: "skip" as const, number: 41762, reason: "still linked to open #41751" };
const commented = { kind: "commented" as const, pullRequest: 41767, tag: "v1.103.0-rc.1", body: fixedBody(41767, { tag: "v1.103.0-rc.1", shipped: false }) };
test("an issue run names the comment, each close, and each pull request left open", () => {
const summary = describeIssue(config, ISSUE, { comment: commented, pullRequests: [closed, left] });
expect(summary).toContain("#41750: commented, fixed by #41767 in v1.103.0-rc.1");
expect(summary).toContain("#41760: closed with: #41750 was fixed by #41767 on main");
expect(summary).toContain("#41762: left open, still linked to open #41751");
expect(summary).not.toContain("DRY RUN");
});
test("a close dry run names the repo variable that turns closing on", () => {
const summary = describeIssue({ ...config, closeDryRun: true }, ISSUE, { comment: commented, pullRequests: [closed] });
expect(summary).toContain("ISSUE_FIXED_CLOSE_PRS_ENABLED");
expect(summary).toContain("#41760: DRY RUN, would close with: #41750 was fixed by #41767 on main");
});
test("a sweep summary counts what it saw and lists only the closes", () => {
const summary = describeSweep(config, { considered: 3522, pullRequests: [closed, left] });
expect(summary).toContain("Swept 3522 open pull requests, 2 linked to an issue, 1 closed");
expect(summary).toContain("#41760: closed with:");
expect(summary).not.toContain("#41762");
expect(describeSweep({ ...config, closeDryRun: true }, { considered: 3522, pullRequests: [closed] })).toContain("1 would be closed, DRY RUN, set the ISSUE_FIXED_CLOSE_PRS_ENABLED");
});
});

View file

@ -6,35 +6,66 @@ declare const process: { readonly env: Readonly<Record<string, string | undefine
export interface FixedConfig {
readonly repo: string;
readonly issueNumber: number;
readonly defaultBranch: string;
readonly dryRun: boolean;
readonly commentDryRun: boolean;
readonly closeDryRun: boolean;
}
export type Run = { readonly kind: "issue"; readonly number: number } | { readonly kind: "sweep" };
interface PullRequestCloser {
readonly __typename: "PullRequest";
readonly number: number;
readonly merged: boolean;
readonly baseRefName: string;
readonly repository: { readonly nameWithOwner: string };
readonly mergeCommit: { readonly oid: string } | null;
}
interface CommitCloser {
readonly __typename: "Commit";
readonly oid: string;
readonly repository: { readonly nameWithOwner: string };
}
export interface ClosedIssue {
export interface IssueClosure {
readonly state: "OPEN" | "CLOSED";
readonly timelineItems: {
readonly nodes: readonly { readonly closer: PullRequestCloser | CommitCloser | null }[];
};
}
export interface LinkedIssue extends IssueClosure {
readonly number: number;
readonly repository: { readonly nameWithOwner: string };
}
export interface LinkedPullRequest {
readonly number: number;
readonly state: "OPEN" | "CLOSED" | "MERGED";
readonly baseRefName: string;
readonly repository: { readonly nameWithOwner: string };
readonly closingIssuesReferences: { readonly totalCount: number; readonly nodes: readonly LinkedIssue[] };
readonly reopens: { readonly nodes: readonly { readonly createdAt: string }[] };
}
export interface PullRequestsPage {
readonly pageInfo: { readonly hasNextPage: boolean; readonly endCursor: string | null };
readonly nodes: readonly LinkedPullRequest[];
}
export interface ClosedIssue extends IssueClosure {
readonly closedByPullRequestsReferences: PullRequestsPage;
}
interface TimelineResponse {
readonly data?: { readonly repository?: { readonly issue: ClosedIssue | null } };
}
interface OpenPullRequestsResponse {
readonly data?: { readonly repository?: { readonly pullRequests: PullRequestsPage } };
}
interface MatchingRef {
readonly ref: string;
}
@ -59,33 +90,100 @@ export type FixedVerdict =
| { readonly kind: "commented"; readonly pullRequest: number; readonly tag: string; readonly body: string }
| { readonly kind: "skip"; readonly reason: string };
export const FIXED_MARKER = "<!-- litellm:fixed-in -->";
const MAX_MINOR_BUMPS = 3;
export type FixSource =
| { readonly kind: "pull_request"; readonly number: number; readonly oid: string }
| { readonly kind: "commit"; readonly oid: string };
export const CLOSER_QUERY = `query($owner: String!, $name: String!, $number: Int!) {
repository(owner: $owner, name: $name) {
issue(number: $number) {
state
timelineItems(last: 1, itemTypes: [CLOSED_EVENT]) {
nodes {
... on ClosedEvent {
closer {
__typename
... on PullRequest { number merged baseRefName mergeCommit { oid } }
... on Commit { oid }
}
}
export type FixVerdict = FixSource | { readonly kind: "skip"; readonly reason: string };
export interface Fix {
readonly issue: number;
readonly source: FixSource;
}
export type CloseVerdict =
| { readonly kind: "candidate"; readonly fixes: readonly Fix[] }
| { readonly kind: "skip"; readonly reason: string };
export type CloseOutcome =
| { readonly kind: "closed"; readonly number: number; readonly body: string }
| { readonly kind: "skip"; readonly number: number; readonly reason: string };
export interface IssueOutcome {
readonly comment: FixedVerdict;
readonly pullRequests: readonly CloseOutcome[];
}
export interface SweepOutcome {
readonly considered: number;
readonly pullRequests: readonly CloseOutcome[];
}
export const FIXED_MARKER = "<!-- litellm:fixed-in -->";
export const SUPERSEDED_MARKER = "<!-- litellm:superseded -->";
const MAX_MINOR_BUMPS = 3;
const MAX_LINKED_PULL_REQUESTS = 50;
const MAX_LINKED_ISSUES = 10;
const SWEEP_PAGE_SIZE = 100;
const CLOSE_PAUSE_MS = 1000;
const WORKFLOW_LOGIN = "github-actions[bot]";
const OPEN_AGAIN = "the issue is open again";
const isReleaseLine = (base: string): boolean => base.startsWith("release/") || base.includes("stable");
const CLOSURE_FRAGMENT = `fragment Closure on Issue {
state
timelineItems(last: 1, itemTypes: [CLOSED_EVENT]) {
nodes {
... on ClosedEvent {
closer {
__typename
... on PullRequest { number merged baseRefName repository { nameWithOwner } mergeCommit { oid } }
... on Commit { oid repository { nameWithOwner } }
}
}
}
}
}`;
const LINKED_PULL_REQUEST_FRAGMENT = `fragment Linked on PullRequest {
number
state
baseRefName
repository { nameWithOwner }
closingIssuesReferences(first: ${MAX_LINKED_ISSUES}) { totalCount nodes { number repository { nameWithOwner } ...Closure } }
reopens: timelineItems(last: 1, itemTypes: [REOPENED_EVENT]) { nodes { ... on ReopenedEvent { createdAt } } }
}`;
export const CLOSER_QUERY = `query($owner: String!, $name: String!, $number: Int!, $after: String) {
repository(owner: $owner, name: $name) {
issue(number: $number) {
...Closure
closedByPullRequestsReferences(first: ${MAX_LINKED_PULL_REQUESTS}, after: $after) {
pageInfo { hasNextPage endCursor }
nodes { ...Linked }
}
}
}
}
${CLOSURE_FRAGMENT}
${LINKED_PULL_REQUEST_FRAGMENT}`;
export const OPEN_PULL_REQUESTS_QUERY = `query($owner: String!, $name: String!, $after: String) {
repository(owner: $owner, name: $name) {
pullRequests(states: OPEN, first: ${SWEEP_PAGE_SIZE}, after: $after) {
pageInfo { hasNextPage endCursor }
nodes { ...Linked }
}
}
}
${CLOSURE_FRAGMENT}
${LINKED_PULL_REQUEST_FRAGMENT}`;
const skip = (reason: string): { readonly kind: "skip"; readonly reason: string } => ({ kind: "skip", reason });
export function closerOf(issue: ClosedIssue, defaultBranch: string): Closer {
export function closerOf(issue: IssueClosure, repo: string, defaultBranch: string): Closer {
if (issue.state !== "CLOSED") {
return skip("the issue is open again");
return skip(OPEN_AGAIN);
}
const closer = issue.timelineItems.nodes[0]?.closer ?? null;
if (closer === null) {
@ -94,6 +192,9 @@ export function closerOf(issue: ClosedIssue, defaultBranch: string): Closer {
if (closer.__typename === "Commit") {
return skip(`closed by commit ${closer.oid.slice(0, 10)}, not by a pull request`);
}
if (closer.repository.nameWithOwner !== repo) {
return skip(`closed by ${closer.repository.nameWithOwner}#${closer.number}, a pull request in another repository`);
}
if (!closer.merged || closer.mergeCommit === null) {
return skip(`closed by #${closer.number}, which is not merged`);
}
@ -121,8 +222,8 @@ async function tagExists(api: GitHubApi, repo: string, tag: string): Promise<boo
return refs.some((ref) => ref.ref === `refs/tags/${tag}`);
}
async function tagContains(api: GitHubApi, repo: string, tag: string, sha: string): Promise<boolean> {
const comparison = await api.request<Comparison>("GET", `/repos/${repo}/compare/${tag}...${sha}`);
async function refContains(api: GitHubApi, repo: string, ref: string, sha: string): Promise<boolean> {
const comparison = await api.request<Comparison>("GET", `/repos/${repo}/compare/${ref}...${sha}`);
return comparison.status === "behind" || comparison.status === "identical";
}
@ -139,7 +240,7 @@ async function firstReleaseWith(
if (!(await tagExists(api, repo, tag))) {
return { kind: "release", tag, shipped: false };
}
if (await tagContains(api, repo, tag, sha)) {
if (await refContains(api, repo, tag, sha)) {
return { kind: "release", tag, shipped: true };
}
if (bumpsLeft === 0) {
@ -164,21 +265,13 @@ export function fixedBody(pullRequest: number, release: { readonly tag: string;
return `${FIXED_MARKER}\nFixed by #${pullRequest}. ${availability}`;
}
export async function commentFixedIssue(api: GitHubApi, config: FixedConfig): Promise<FixedVerdict> {
const [owner, name] = config.repo.split("/");
const response = await api.request<TimelineResponse>("POST", "/graphql", {
query: CLOSER_QUERY,
variables: { owner, name, number: config.issueNumber },
});
const issue = response.data?.repository?.issue ?? null;
if (issue === null) {
return skip("not an issue in this repository");
}
const closer = closerOf(issue, config.defaultBranch);
if (closer.kind === "skip") {
return closer;
}
const issuePath = `/repos/${config.repo}/issues/${config.issueNumber}`;
export async function commentFixedIssue(
api: GitHubApi,
config: FixedConfig,
issueNumber: number,
closer: { readonly number: number; readonly mergeCommit: string },
): Promise<FixedVerdict> {
const issuePath = `/repos/${config.repo}/issues/${issueNumber}`;
const comments = await listAll<Comment>(api, `${issuePath}/comments`);
if (comments.some((comment) => comment.body.includes(FIXED_MARKER))) {
return skip("already carries a fixed-in comment");
@ -188,37 +281,259 @@ export async function commentFixedIssue(api: GitHubApi, config: FixedConfig): Pr
return release;
}
const body = fixedBody(closer.number, release);
if (!config.dryRun) {
if (!config.commentDryRun) {
await api.request("POST", `${issuePath}/comments`, { body });
}
return { kind: "commented", pullRequest: closer.number, tag: release.tag, body };
}
export function readConfig(env: Readonly<Record<string, string | undefined>>): FixedConfig & { readonly token: string } {
export function fixOf(issue: IssueClosure, repo: string): FixVerdict {
if (issue.state !== "CLOSED") {
return skip("is open again");
}
const closer = issue.timelineItems.nodes[0]?.closer ?? null;
if (closer === null) {
return skip("was closed by hand");
}
if (closer.repository.nameWithOwner !== repo) {
return skip(`was closed from ${closer.repository.nameWithOwner}`);
}
if (closer.__typename === "Commit") {
return { kind: "commit", oid: closer.oid };
}
if (!closer.merged || closer.mergeCommit === null) {
return skip(`was closed by #${closer.number}, which is not merged`);
}
return { kind: "pull_request", number: closer.number, oid: closer.mergeCommit.oid };
}
export function closeVerdict(pullRequest: LinkedPullRequest, config: FixedConfig): CloseVerdict {
if (pullRequest.repository.nameWithOwner !== config.repo) {
return skip(`lives in ${pullRequest.repository.nameWithOwner}`);
}
if (pullRequest.state !== "OPEN") {
return skip(`is ${pullRequest.state.toLowerCase()}`);
}
if (isReleaseLine(pullRequest.baseRefName)) {
return skip(`targets the release line ${pullRequest.baseRefName}`);
}
const { totalCount, nodes: linked } = pullRequest.closingIssuesReferences;
if (linked.length === 0) {
return skip("links no issue");
}
if (totalCount > linked.length) {
return skip(`links ${totalCount} issues, more than the ${MAX_LINKED_ISSUES} this workflow reads`);
}
const foreign = linked.find((issue) => issue.repository.nameWithOwner !== config.repo);
if (foreign !== undefined) {
return skip(`links ${foreign.repository.nameWithOwner}#${foreign.number}`);
}
const stillOpen = linked.find((issue) => issue.state === "OPEN");
if (stillOpen !== undefined) {
return skip(`still linked to open #${stillOpen.number}`);
}
const verdicts = linked.map((issue) => ({ issue: issue.number, fix: fixOf(issue, config.repo) }));
for (const { issue, fix } of verdicts) {
if (fix.kind === "skip") {
return skip(`#${issue} ${fix.reason}`);
}
}
return {
kind: "candidate",
fixes: verdicts.flatMap(({ issue, fix }) => (fix.kind === "skip" ? [] : [{ issue, source: fix }])),
};
}
function describeSource(source: FixSource): string {
return source.kind === "pull_request" ? `#${source.number}` : `commit ${source.oid.slice(0, 10)}`;
}
async function fixOffDefaultBranch(api: GitHubApi, config: FixedConfig, fixes: readonly Fix[]): Promise<Fix | undefined> {
const onBranch = await Promise.all(fixes.map((fix) => refContains(api, config.repo, config.defaultBranch, fix.source.oid)));
return fixes.find((_, index) => !onBranch[index]);
}
export function supersededBody(fixes: readonly Fix[], defaultBranch: string): string {
const pairs = fixes
.map((fix, index) => `#${fix.issue} ${index === 0 ? "was fixed by" : "by"} ${describeSource(fix.source)}`)
.join(" and ");
return `${SUPERSEDED_MARKER}\n${pairs} on ${defaultBranch}, so this pull request is closed. Reopen it if something was missed.`;
}
function reopenedAfter(pullRequest: LinkedPullRequest, comment: Comment): boolean {
const reopen = pullRequest.reopens.nodes[0];
return reopen !== undefined && Date.parse(reopen.createdAt) > Date.parse(comment.created_at);
}
async function closePullRequest(
api: GitHubApi,
config: FixedConfig,
pullRequest: LinkedPullRequest,
pause: () => Promise<void>,
): Promise<CloseOutcome> {
const verdict = closeVerdict(pullRequest, config);
if (verdict.kind === "skip") {
return { kind: "skip", number: pullRequest.number, reason: verdict.reason };
}
const offBranch = await fixOffDefaultBranch(api, config, verdict.fixes);
if (offBranch !== undefined) {
const fix = describeSource(offBranch.source);
return { kind: "skip", number: pullRequest.number, reason: `#${offBranch.issue} was fixed by ${fix}, which is not on ${config.defaultBranch}` };
}
const issuePath = `/repos/${config.repo}/issues/${pullRequest.number}`;
const comments = await listAll<Comment>(api, `${issuePath}/comments`);
const marker = comments.find((comment) => comment.user.login === WORKFLOW_LOGIN && comment.body.includes(SUPERSEDED_MARKER));
if (marker !== undefined && reopenedAfter(pullRequest, marker)) {
return { kind: "skip", number: pullRequest.number, reason: "was closed by this workflow once and reopened" };
}
const body = marker?.body ?? supersededBody(verdict.fixes, config.defaultBranch);
if (config.closeDryRun) {
return { kind: "closed", number: pullRequest.number, body };
}
if (marker === undefined) {
await pause();
await api.request("POST", `${issuePath}/comments`, { body });
}
await pause();
await api.request("PATCH", `/repos/${config.repo}/pulls/${pullRequest.number}`, { state: "closed" });
return { kind: "closed", number: pullRequest.number, body };
}
export function closePullRequests(
api: GitHubApi,
config: FixedConfig,
candidates: readonly LinkedPullRequest[],
pause: () => Promise<void>,
): Promise<readonly CloseOutcome[]> {
return candidates.reduce<Promise<readonly CloseOutcome[]>>(
async (previous, candidate) => [...(await previous), await closePullRequest(api, config, candidate, pause)],
Promise.resolve([]),
);
}
type NextPage = (after: string | null) => Promise<PullRequestsPage>;
async function collectPages(page: PullRequestsPage, nextPage: NextPage): Promise<readonly LinkedPullRequest[]> {
if (!page.pageInfo.hasNextPage) {
return page.nodes;
}
return [...page.nodes, ...(await collectPages(await nextPage(page.pageInfo.endCursor), nextPage))];
}
async function closedIssue(api: GitHubApi, config: FixedConfig, issueNumber: number, after: string | null): Promise<ClosedIssue | null> {
const [owner, name] = config.repo.split("/");
const response = await api.request<TimelineResponse>("POST", "/graphql", {
query: CLOSER_QUERY,
variables: { owner, name, number: issueNumber, after },
});
return response.data?.repository?.issue ?? null;
}
export async function handleFixedIssue(
api: GitHubApi,
config: FixedConfig,
issueNumber: number,
pause: () => Promise<void>,
): Promise<IssueOutcome> {
const issue = await closedIssue(api, config, issueNumber, null);
if (issue === null) {
return { comment: skip("not an issue in this repository"), pullRequests: [] };
}
if (issue.state !== "CLOSED") {
return { comment: skip(OPEN_AGAIN), pullRequests: [] };
}
const closer = closerOf(issue, config.repo, config.defaultBranch);
const comment = closer.kind === "skip" ? closer : await commentFixedIssue(api, config, issueNumber, closer);
const nextPage: NextPage = async (after) => {
const more = await closedIssue(api, config, issueNumber, after);
if (more === null) {
throw new Error(`#${issueNumber} came back without data while reading its linked pull requests after cursor ${after}`);
}
return more.closedByPullRequestsReferences;
};
const linked = await collectPages(issue.closedByPullRequestsReferences, nextPage);
const open = linked.filter((pullRequest) => pullRequest.state === "OPEN");
const pullRequests = await closePullRequests(api, config, open, pause);
return { comment, pullRequests };
}
async function openPullRequests(api: GitHubApi, config: FixedConfig): Promise<readonly LinkedPullRequest[]> {
const [owner, name] = config.repo.split("/");
const nextPage: NextPage = async (after) => {
const response = await api.request<OpenPullRequestsResponse>("POST", "/graphql", {
query: OPEN_PULL_REQUESTS_QUERY,
variables: { owner, name, after },
});
const page = response.data?.repository?.pullRequests;
if (page === undefined) {
throw new Error(`open pull requests after cursor ${after} came back without data: ${JSON.stringify(response)}`);
}
return page;
};
return collectPages(await nextPage(null), nextPage);
}
export async function sweep(api: GitHubApi, config: FixedConfig, pause: () => Promise<void>): Promise<SweepOutcome> {
const open = await openPullRequests(api, config);
const linked = open.filter((pullRequest) => pullRequest.closingIssuesReferences.nodes.length > 0);
return { considered: open.length, pullRequests: await closePullRequests(api, config, linked, pause) };
}
export function readConfig(
env: Readonly<Record<string, string | undefined>>,
): FixedConfig & { readonly token: string; readonly run: Run } {
const token = env.GITHUB_TOKEN;
const repo = env.GITHUB_REPOSITORY;
const defaultBranch = env.DEFAULT_BRANCH;
if (!token || !repo || !/^[\w.-]+\/[\w.-]+$/.test(repo) || !defaultBranch) {
throw new Error("GITHUB_TOKEN, GITHUB_REPOSITORY (owner/repo) and DEFAULT_BRANCH are required");
}
const config = { token, repo, defaultBranch, commentDryRun: env.DRY_RUN === "true", closeDryRun: env.CLOSE_PRS_DRY_RUN === "true" };
if (env.SWEEP === "true") {
return { ...config, run: { kind: "sweep" } };
}
const issueNumber = Number(env.ISSUE_NUMBER);
if (!Number.isInteger(issueNumber) || issueNumber <= 0) {
throw new Error(`ISSUE_NUMBER must be a positive integer, got "${env.ISSUE_NUMBER}"`);
throw new Error(`ISSUE_NUMBER must be a positive integer, got "${env.ISSUE_NUMBER}": dispatch with an issue_number or with sweep ticked`);
}
return { token, repo, issueNumber, defaultBranch, dryRun: env.DRY_RUN === "true" };
return { ...config, run: { kind: "issue", number: issueNumber } };
}
function describe(config: FixedConfig, verdict: FixedVerdict): string {
if (verdict.kind === "skip") {
return `#${config.issueNumber}: skipped, ${verdict.reason}`;
const CLOSE_DRY_RUN_HINT = "set the ISSUE_FIXED_CLOSE_PRS_ENABLED repo variable to true to close pull requests";
function describeClose(config: FixedConfig, outcome: CloseOutcome): string {
if (outcome.kind === "skip") {
return `#${outcome.number}: left open, ${outcome.reason}`;
}
if (config.dryRun) {
return `#${config.issueNumber}: DRY RUN, set the ISSUE_FIXED_COMMENT_ENABLED repo variable to true to post this:\n\n${verdict.body}`;
}
return `#${config.issueNumber}: commented, fixed by #${verdict.pullRequest} in ${verdict.tag}`;
const text = outcome.body.replace(`${SUPERSEDED_MARKER}\n`, "");
return config.closeDryRun ? `#${outcome.number}: DRY RUN, would close with: ${text}` : `#${outcome.number}: closed with: ${text}`;
}
export function describeIssue(config: FixedConfig, issueNumber: number, outcome: IssueOutcome): string {
const comment =
outcome.comment.kind === "skip"
? `#${issueNumber}: skipped, ${outcome.comment.reason}`
: config.commentDryRun
? `#${issueNumber}: DRY RUN, set the ISSUE_FIXED_COMMENT_ENABLED repo variable to true to post this:\n\n${outcome.comment.body}`
: `#${issueNumber}: commented, fixed by #${outcome.comment.pullRequest} in ${outcome.comment.tag}`;
const hint = config.closeDryRun && outcome.pullRequests.some((pullRequest) => pullRequest.kind === "closed") ? [`Closing is a DRY RUN, ${CLOSE_DRY_RUN_HINT}`] : [];
return [comment, ...hint, ...outcome.pullRequests.map((pullRequest) => describeClose(config, pullRequest))].join("\n");
}
export function describeSweep(config: FixedConfig, outcome: SweepOutcome): string {
const closed = outcome.pullRequests.filter((pullRequest) => pullRequest.kind === "closed");
const verb = config.closeDryRun ? `would be closed, DRY RUN, ${CLOSE_DRY_RUN_HINT}` : "closed";
const header = `Swept ${outcome.considered} open pull requests, ${outcome.pullRequests.length} linked to an issue, ${closed.length} ${verb}`;
return [header, ...closed.map((pullRequest) => describeClose(config, pullRequest))].join("\n");
}
if (import.meta.main) {
const { token, ...config } = readConfig(process.env);
console.log(describe(config, await commentFixedIssue(githubApi(token), config)));
const { token, run, ...config } = readConfig(process.env);
const api = githubApi(token);
const pause = (): Promise<void> => new Promise((resolve) => setTimeout(resolve, CLOSE_PAUSE_MS));
console.log(
run.kind === "sweep"
? describeSweep(config, await sweep(api, config, pause))
: describeIssue(config, run.number, await handleFixedIssue(api, config, run.number, pause)),
);
}

View file

@ -69,6 +69,9 @@ IGNORE_FUNCTIONS = [
"_unqualified", # bounded by the qualifier depth of a static TypedDict annotation (Annotated, Required/NotRequired, ReadOnly around one type, no cycles possible).
"_render_json", # bounded by the nesting depth of a pydantic-validated JsonValue from the operator's config (a finite JSON tree, no cycles possible).
"completion_cost", # max depth 1: recursion only fires for mixed-tier Responses WS logging objects, and each split part carries a single service_tier so _split_responses_ws_logging_object_by_service_tier returns None.
"_string_leaves", # bounded by the nesting depth of a safe_json_structure output (a finite JSON tree, no cycles possible).
"_replace_string_leaves", # bounded by the nesting depth of a safe_json_structure output (a finite JSON tree, no cycles possible).
"_sort_processed_sets", # bounded by the nesting depth of the log-record extra it walks (a finite JSON tree, no cycles possible).
]

View file

@ -0,0 +1,41 @@
FROM debian:bookworm-slim@sha256:3783cc01769c7b2b1b83a5c5ad96c815348e28ed7da68e2e3687004faa906251
ARG GH_VERSION=2.101.0
ARG GH_SHA256=9bca2d1c16825f109907a23307628a2f0698fbf99662b73a5cf0b020293072b8
ARG UV_VERSION=0.10.9
ARG UV_SHA256=20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594
ARG CLAUDE_CODE_VERSION=2.1.228
ARG CLAUDE_CODE_SHA256=d535985e6941a3eb00179ccd7f52ceb0c6623a0305a518ebc4e6514f84a94c99
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates curl git jq procps iproute2 \
&& rm -rf /var/lib/apt/lists/*
RUN curl -fsSLo /tmp/gh.tar.gz "https://github.com/cli/cli/releases/download/v${GH_VERSION}/gh_${GH_VERSION}_linux_amd64.tar.gz" \
&& echo "${GH_SHA256} /tmp/gh.tar.gz" | sha256sum -c - \
&& tar -xzf /tmp/gh.tar.gz -C /usr/local/bin --strip-components=2 "gh_${GH_VERSION}_linux_amd64/bin/gh" \
&& rm /tmp/gh.tar.gz
RUN curl -fsSLo /tmp/uv.tar.gz "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz" \
&& echo "${UV_SHA256} /tmp/uv.tar.gz" | sha256sum -c - \
&& tar -xzf /tmp/uv.tar.gz -C /usr/local/bin --strip-components=1 uv-x86_64-unknown-linux-gnu/uv \
&& rm /tmp/uv.tar.gz
RUN curl -fsSLo /tmp/claude "https://downloads.claude.ai/claude-code-releases/${CLAUDE_CODE_VERSION}/linux-x64/claude" \
&& echo "${CLAUDE_CODE_SHA256} /tmp/claude" | sha256sum -c - \
&& install -m 0755 /tmp/claude /usr/local/bin/claude \
&& rm /tmp/claude
RUN groupadd --gid 1000 populator && useradd --uid 1000 --gid 1000 --create-home populator
ENV HOME=/home/populator \
LITELLM_REPO=/opt/litellm \
DISABLE_AUTOUPDATER=1
COPY --chown=populator:populator . /opt/litellm/tests/e2e/
USER populator
WORKDIR /home/populator
CMD ["/opt/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh"]

View file

@ -1,66 +1,59 @@
# Cron VM setup for the Claude Code compatibility-matrix populator
# Render cron job for the Claude Code compatibility-matrix populator
The populator runs daily on a dedicated GCP VM
(`litellm-compatibility-matrix-populator`) rather than as a GitHub
Action. Trade-offs:
The populator runs daily as the Render cron job `litellm-compat-matrix`
(Docker runtime, built from the `Dockerfile` in this directory) rather
than as a GitHub Action or on a dedicated VM. Trade-offs:
- ✅ Real VM means we can `gh auth login` against an account that's
already a collaborator on `BerriAI/litellm-docs`, instead of
provisioning a GitHub App with `pull-requests: write`.
- ✅ Persistent state (a single `~/litellm-cron-worktree/` and its `.venv`)
is reused across runs, so each daily run does a fast `git checkout` +
incremental `uv sync` rather than a fresh clone + cold sync.
- ✅ No Docker dependency — the proxy runs directly via `uv run litellm`.
- ⚠️ The VM has to actually be on. systemd's `Persistent=true` recovers
from short outages, but a multi-day outage means the matrix goes
stale until the VM is back.
- ⚠️ Provider credentials live on the VM filesystem
(`/etc/litellm-compat-matrix.env`) instead of GitHub secrets. Treat
the VM as an environment with comparable blast radius to a CI runner.
This directory used to live at `tests/claude_code/cron_vm/` (paired with
the standalone `tests/claude_code/` suite); it now runs the maintained
`tests/e2e/claude_code/` suite instead. The pytest env interface changed
accordingly: the runner exports `LITELLM_PROXY_URL` / `LITELLM_MASTER_KEY`
(previously `LITELLM_PROXY_BASE_URL` / `LITELLM_PROXY_API_KEY`), the azure
column reads `AZURE_AI_API_KEY` / `AZURE_AI_API_BASE` (previously
`AZURE_FOUNDRY_*`), and the GPT columns need `OPENAI_API_KEY` and
`AZURE_API_BASE` / `AZURE_API_KEY` — see `litellm-compat-matrix.env.example`.
- ✅ No machine to keep on or patch. Render builds the image from this
directory on every push to `main` that touches `tests/e2e/**` and
runs it on the schedule.
- ✅ Credentials live in Render env vars and secret files, scoped to
this one service, instead of on a VM filesystem.
- ✅ The publish token still uses the `mateo-berri` account, which is a
collaborator on `BerriAI/litellm-docs`, so no GitHub App with
`pull-requests: write` has to be provisioned.
- ⚠️ The disk is ephemeral, so every run starts from a fresh (blobless)
clone of litellm plus a cold `uv sync`. That adds a few minutes on
top of the ~10 minute test run; the job's 12 hour ceiling is nowhere
near.
- ⚠️ The Claude Code CLI version under test is pinned in the
`Dockerfile` (`CLAUDE_CODE_VERSION` + its checksum). Bumping it is a
PR, see the gotchas below.
## Layout
| File | Purpose |
| --- | --- |
| `run_daily.sh` | The actual cron job. Resolves versions, updates the worktree, boots the proxy, runs pytest, builds the JSON, opens (or updates) a docs PR, sweeps stale compat-matrix PRs. |
| `Dockerfile` | The image Render builds: Debian bookworm-slim plus pinned, checksum-verified `gh`, `uv`, and the Claude Code CLI, with this `tests/e2e/` tree copied to `/opt/litellm/tests/e2e/`. Runs as the non-root user `populator` (uid/gid 1000, which is what Render's secret files are readable by). |
| `run_daily.sh` | The actual cron job. Resolves versions, clones the worktree, boots the proxy, runs pytest, builds the JSON, opens (or updates) a docs PR, sweeps stale compat-matrix PRs. |
| `build_matrix.py` | Tiny Python CLI that wraps `claude_code.matrix_builder.build_from_paths`. Exists only because the bash script needs *some* way to render the per-cell aggregation, and the builder is already Python. |
| `check_regressions.py` | Tiny Python CLI that wraps `claude_code.matrix_builder.find_regressions`. Diffs the freshly built matrix against the currently-published one and exits `3` if any cell flipped green→red, which gates auto-merge. |
| `litellm-compat-matrix.service` | systemd oneshot that invokes `run_daily.sh`. |
| `litellm-compat-matrix.timer` | `OnCalendar=*-*-* 06:00:00 UTC`, `Persistent=true`. |
| `litellm-compat-matrix.env.example` | Template for `/etc/litellm-compat-matrix.env`. |
| `litellm-compat-matrix.env.example` | The service's env vars, one per line, with what each is for. |
## What `run_daily.sh` does
1. **Resolves the latest LiteLLM final release tag** (newest bare
`vX.Y.Z`, skipping `-rc.N`/`-dev.N` pre-releases) by paging the
GitHub Releases API (`curl | jq`).
2. **Reads the local Claude Code CLI version** via `claude --version`.
The cron does not auto-upgrade the CLI — operators do that
out-of-band by running `npm install -g @anthropic-ai/claude-code@latest`.
3. **Updates the persistent worktree** at `~/litellm-cron-worktree/`:
`git fetch --tags --force`, `git reset --hard`,
`git clean -fdx -e .venv -e .uv-bin`, `git checkout --force <tag>`.
The `.venv` is preserved across runs so `uv sync --frozen` is
incremental. Then **shims the test suite**: `tests/e2e/` in the
worktree is rebuilt from the dev checkout — the `claude_code/` suite
plus the five shared transport helpers it imports (`proxy_client.py`,
`e2e_http.py`, `models.py`, `e2e_config.py`, `transport.py`) — so the
cron always runs *today's* tests against the latest stable proxy. The
tag's own `tests/e2e/` tree (including the EKS-harness `conftest.py`,
whose imports the stable venv doesn't install) is deliberately not
used.
2. **Reads the Claude Code CLI version** via `claude --version`. That
is whatever the `Dockerfile` pins; the job never upgrades it on its
own.
3. **Clones the worktree** at `~/litellm-cron-worktree/` (a
`--filter=blob:none` clone, so only the checked-out tag's blobs are
fetched), `git checkout --force <tag>`, then `uv sync --frozen
--no-install-project` against a uv-managed CPython 3.12 followed by
`uv pip install --no-build litellm==<version>`, so the proxy under
test is the published PyPI wheel (what users install) rather than a
source build: the tag builds a Rust extension through maturin, and
the image ships no C or Rust toolchain. Then **shims the test suite**:
`tests/e2e/` in the worktree is replaced by the image's copy of this
whole tree, so the cron always runs *today's* tests against the
latest stable proxy, and pytest runs with `--confcutdir` pointed at
`claude_code/` so the tree's EKS-harness `conftest.py` (whose imports
the stable venv doesn't install) is never loaded. The tag's own
`tests/e2e/` is deliberately not used.
4. **Boots the proxy** as a `setsid` background process on port `4100`
(so it can't collide with a developer's `:4000`), then polls
`/health/liveliness` until it's up.
bound to loopback, then polls `/health/liveliness` until it's up.
5. **Runs pytest** on `tests/e2e/claude_code/` with `LITELLM_PROXY_URL`
pointed at the proxy and `COMPAT_RESULTS_PATH` set so the conftest
hook writes the per-test results artifact. Test failures become
@ -74,8 +67,9 @@ column reads `AZURE_AI_API_KEY` / `AZURE_AI_API_BASE` (previously
`mateo-berri` token has write access, so this is a same-repo branch,
not a fork), `gh pr create`. A re-run on the same day fast-forwards
the existing branch and `gh pr create` no-ops ("a pull request for
branch ... already exists" is treated as success). These PRs are no
longer gated on a second human review.
branch ... already exists" is treated as success). If the JSON is
byte-identical to what `main` already publishes, the push is skipped
entirely. These PRs are not gated on a second human review.
8. **Gates auto-merge on a regression check**: before enabling
auto-merge, `check_regressions.py` diffs the new matrix against the
one currently on `main`. Auto-merge (`gh pr merge --auto --squash`)
@ -89,107 +83,155 @@ column reads `AZURE_AI_API_KEY` / `AZURE_AI_API_BASE` (previously
human reviews before it lands on the public table. The check fails
*closed*: if it errors, auto-merge is withheld.
9. **Sweeps stale compat-matrix PRs**: once today's PR exists, every
other open `compat-matrix/*` PR on the docs repo is closed (and its
bot-owned branch deleted), so at most one compat-matrix PR is ever
open — the newest.
other open `compat-matrix/*` PR that the publishing account opened
from a branch on the docs repo itself is closed (and its bot-owned
branch deleted), so at most one compat-matrix PR is ever open — the
newest. A contributor's PR under that prefix is never touched.
## One-time VM setup
## The Render service
Run as `mateo` on the cron VM:
Everything below is what the live service is set to; recreate it with
the same values if it ever has to be rebuilt.
| Setting | Value |
| --- | --- |
| Workspace | Litellm (the one that already builds the other litellm services) |
| Type | Cron job, Docker runtime |
| Repo / branch | `BerriAI/litellm` @ `main` |
| Dockerfile path | `tests/e2e/claude_code/cron_vm/Dockerfile` |
| Docker build context | `tests/e2e` (the repo root `.dockerignore` excludes `tests`, so the context has to start below it) |
| Build filter | included paths `tests/e2e/**` |
| Schedule | `0 6 * * *` (06:00 UTC daily) |
| Plan / region | `4c-16g` (4 CPU, 16 GB, what the dashboard calls Pro Max; the suite fans out to ~75 concurrent CLI calls) / Oregon |
| Env vars | every key in `litellm-compat-matrix.env.example` |
| Secret files | `github-token` (the publish PAT, one line) and `vertex-service-account.json` (the Vertex service-account key) |
Render mounts secret files at `/etc/secrets/<name>`, which is where
`CREDENTIALS_DIRECTORY` and `GOOGLE_APPLICATION_CREDENTIALS` in the env
example point. Render also passes env vars to `docker build` as build
args, which is why the `Dockerfile` declares no `ARG` that could ever
be given a secret's name.
Creating it through the API looks like this (fill `envVars` and
`secretFiles` from the env example and the two secrets; `ownerId` is
the workspace id from `GET /v1/owners`):
```bash
# 1. Toolchain
sudo apt-get update
sudo apt-get install -y git nodejs npm jq curl
curl -LsSf https://astral.sh/uv/install.sh | sh
sudo apt-get install -y gh # or follow https://cli.github.com/
# 2. Claude Code CLI (the cron does NOT auto-upgrade this; rerun this
# line out-of-band when you want a fresh CLI to be tested)
sudo npm install -g @anthropic-ai/claude-code@latest
# 3. Litellm checkout. Used by systemd's WorkingDirectory and as the
# source of the .service / .timer files. The cron itself runs out
# of the separate worktree at ~/litellm-cron-worktree/.
mkdir -p ~/litellm
git clone https://github.com/BerriAI/litellm.git ~/litellm/litellm
git -C ~/litellm/litellm checkout litellm_internal_staging
# 4. gh auth — must be a collaborator on BerriAI/litellm-docs.
gh auth login # follow prompts; pick HTTPS + token paste flow
# 5. Provider credentials + the publish token.
sudo cp ~/litellm/litellm/tests/e2e/claude_code/cron_vm/litellm-compat-matrix.env.example \
/etc/litellm-compat-matrix.env
sudoedit /etc/litellm-compat-matrix.env # fill in real values
sudo chmod 0600 /etc/litellm-compat-matrix.env
# The mateo-berri PAT lives in its own file, mapped into the service via
# systemd LoadCredential so it stays out of the test processes' env
# (see the env.example comment for why).
sudo install -m 0600 /dev/null /etc/litellm-compat-matrix-github-token
sudoedit /etc/litellm-compat-matrix-github-token # single line: the PAT
# 6. systemd units.
sudo cp ~/litellm/litellm/tests/e2e/claude_code/cron_vm/litellm-compat-matrix.service /etc/systemd/system/
sudo cp ~/litellm/litellm/tests/e2e/claude_code/cron_vm/litellm-compat-matrix.timer /etc/systemd/system/
sudo systemctl daemon-reload
sudo systemctl enable --now litellm-compat-matrix.timer
curl -fsS https://api.render.com/v1/services \
-H "Authorization: Bearer ${RENDER_API_KEY}" \
-H 'Content-Type: application/json' \
-d '{
"type": "cron_job",
"name": "litellm-compat-matrix",
"ownerId": "<workspace id>",
"repo": "https://github.com/BerriAI/litellm",
"branch": "main",
"autoDeploy": "yes",
"buildFilter": {"paths": ["tests/e2e/**"], "ignoredPaths": []},
"envVars": [{"key": "ANTHROPIC_API_KEY", "value": "..."}],
"secretFiles": [{"name": "github-token", "content": "..."},
{"name": "vertex-service-account.json", "content": "..."}],
"serviceDetails": {
"runtime": "docker",
"schedule": "0 6 * * *",
"plan": "4c-16g",
"region": "oregon",
"envSpecificDetails": {
"dockerfilePath": "tests/e2e/claude_code/cron_vm/Dockerfile",
"dockerContext": "tests/e2e"
}
}
}'
```
## Operating it
```bash
# When does it run next?
systemctl list-timers litellm-compat-matrix.timer
# Trigger a real run right now (PRs to litellm-docs). The id is the
# service id (`crn-...`) from the dashboard URL or `GET /v1/services`.
curl -fsS -X POST "https://api.render.com/v1/cron-jobs/${CRON_ID}/runs" \
-H "Authorization: Bearer ${RENDER_API_KEY}"
# Trigger a real run right now (PRs to litellm-docs).
sudo systemctl start litellm-compat-matrix.service
# Follow a run: the Logs tab on the service, or the API.
curl -fsS "https://api.render.com/v1/logs?ownerId=${OWNER_ID}&resource=${CRON_ID}&limit=100" \
-H "Authorization: Bearer ${RENDER_API_KEY}"
# Trigger a run that does NOT open a PR (good for first-time validation).
SKIP_PUBLISH=1 ~/litellm/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh
# Rebuild the image after a merge that touches tests/e2e/** (see the
# auto-deploy gotcha below). The deploy is done once its status is
# `live`; a run triggered before that still uses the previous image.
curl -fsS -X POST "https://api.render.com/v1/services/${CRON_ID}/deploys" \
-H "Authorization: Bearer ${RENDER_API_KEY}" \
-H 'Content-Type: application/json' -d '{"clearCache": "do_not_clear"}'
curl -fsS "https://api.render.com/v1/services/${CRON_ID}/deploys?limit=1" \
-H "Authorization: Bearer ${RENDER_API_KEY}"
# Narrow to one cell while debugging.
SKIP_PUBLISH=1 PYTEST_K='basic_messaging_non_streaming and anthropic' \
~/litellm/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh
# A run that does NOT open a PR (first-time validation, CLI bumps):
# set SKIP_PUBLISH=1 on the service, trigger a run, then remove it.
# The matrix JSON is printed at the end of the run's log (nothing on
# the container's disk outlives the run) and saved to
# ~/compatibility-matrix.json for a local docker run.
# PYTEST_K='basic_messaging_non_streaming and anthropic' narrows the
# run to one cell the same way.
# Watch the most recent run.
journalctl -u litellm-compat-matrix.service -f
# Read older runs.
journalctl -u litellm-compat-matrix.service --since '2 days ago'
# Disable until further notice (e.g. while debugging).
sudo systemctl disable --now litellm-compat-matrix.timer
# Build and run the image locally (docker on Apple silicon needs the
# platform flag; the context is tests/e2e, see the table above).
docker build --platform linux/amd64 \
-f tests/e2e/claude_code/cron_vm/Dockerfile -t compat-matrix tests/e2e
docker run --rm --platform linux/amd64 \
--env-file litellm-compat-matrix.env -e SKIP_PUBLISH=1 \
-v "$PWD/secrets:/etc/secrets:ro" compat-matrix
```
## Gotchas
- **The venv is pinned to Python 3.12 (`CRON_PYTHON_VERSION`).** The
e2e suite uses PEP 695 `type` aliases, which the VM's system Python
(3.11) can't parse; `run_daily.sh` has uv fetch a managed CPython
e2e suite uses PEP 695 `type` aliases, which the image's Debian
Python can't parse; `run_daily.sh` has uv fetch a managed CPython
into `~/litellm-cron-worktree/.uv-python/` and syncs the venv against
it. The first run after a version bump is a cold venv rebuild.
- **The proxy port is `4100`, not `4000`.** This is so a developer SSH'd
into the same VM with their own `:4000` proxy doesn't collide with a
cron run. Override with `PROXY_PORT=...` in `/etc/litellm-compat-matrix.env`
if you need to.
it.
- **The proxy port is `4100`, not `4000`.** Kept from the VM days so a
developer running the script locally next to their own `:4000` proxy
doesn't collide. Override with `PROXY_PORT=...`.
- **`uv sync --frozen` requires the resolved tag to be tagged on
GitHub.** If the latest stable release was made but not pushed as a
git tag, the `git checkout` step fails. Push the tag, then rerun.
GitHub, and the wheel install requires it on PyPI.** If the latest
stable release was made but not pushed as a git tag, the `git
checkout` step fails; push the tag, then rerun. PyPI has had every
stable version days before its GitHub release so far (1.102.0 was
uploaded 2026-09-20, released on GitHub 2026-09-22), so the
`--no-build` install failing means the wheel is genuinely missing,
not late.
- **Pushes do not redeploy the service; deploy by hand.** `autoDeploy`
is `yes` on the service, but Render only hears about pushes through
its GitHub app, which is not installed on the `BerriAI` org (an org
admin step), so no push to the branch has ever started a deploy.
After a merge that changes anything under `tests/e2e/**`, run the
deploy command from the operating section (or "Manual Deploy" on the
dashboard) and wait for `live` before triggering a run, otherwise
the next scheduled run still executes the old image.
- **Publish-token rotation is your problem.** The cron does not
refresh the token; if `mateo-berri`'s PAT in
`/etc/litellm-compat-matrix-github-token` expires, the run fails at
the `git push`/`gh pr create` step with a 401 ("Bad credentials" /
"Authentication failed"). Mint a fresh PAT and update that file.
The token needs write access to `BerriAI/litellm-docs` (classic
`repo` scope, or fine-grained Contents:RW + Pull requests:RW). It is
delivered via systemd `LoadCredential`, not the env file, so pytest,
the proxy, and the claude CLI never inherit it; manual runs export
`GITHUB_TOKEN` instead.
- **First run after upgrading the Claude Code CLI is the riskiest one.**
If the new CLI changes its wire format the matrix run can produce
systematic failures. Always run with `SKIP_PUBLISH=1` after a CLI
upgrade before letting the next scheduled fire happen.
- **Disk:** the worktree's `.venv` is ~1.3 GB and the `.git` directory
is ~1 GB. Plan for at least 5 GB free on the VM, otherwise
`uv sync` will fail mid-run and leave you with a half-installed venv.
refresh the token; if `mateo-berri`'s PAT in the `github-token`
secret file expires, the run fails at the `git push`/`gh pr create`
step with a 401 ("Bad credentials" / "Authentication failed"). Mint
a fresh PAT and replace the secret file on the service. The token
needs write access to `BerriAI/litellm-docs` (classic `repo` scope,
or fine-grained Contents:RW + Pull requests:RW). It is delivered as
a file, not an env var, so pytest, the proxy, and the claude CLI
never inherit it; manual runs export `GITHUB_TOKEN` instead.
- **Bumping the Claude Code CLI is a PR.** Change `CLAUDE_CODE_VERSION`
in the `Dockerfile` and set `CLAUDE_CODE_SHA256` to the `linux-x64`
checksum from
`https://downloads.claude.ai/claude-code-releases/<version>/manifest.json`.
The first run on a new CLI is the riskiest one: if the new CLI
changes its wire format the matrix run can produce systematic
failures, so trigger a `SKIP_PUBLISH=1` run before the next scheduled
fire. `gh` and `uv` bump the same way, with the checksum from the
release's `gh_<version>_checksums.txt` and the tarball's `.sha256`
sidecar respectively.
- **A local build on Apple silicon only proves the image assembles.**
Under QEMU the Claude Code binary (a Bun executable) dies with
`CPU lacks AVX support` and `gh` panics in the Go runtime, so
`claude --version` and a full run are verified with a
`SKIP_PUBLISH=1` run on Render, not locally.
- **Nothing persists between runs.** A failed run leaves no
half-installed venv behind, but also no cache: don't expect a rerun
to be faster than the first one.

View file

@ -1,9 +1,8 @@
# Environment file consumed by `litellm-compat-matrix.service`.
# Environment variables of the Render cron job `litellm-compat-matrix`.
#
# Install at `/etc/litellm-compat-matrix.env` and chmod 0600.
# `EnvironmentFile=-` in the unit means the service is allowed to start
# even if this file is missing, but the populator will fail at the
# first provider request without these credentials.
# Set every value here on the Render service (Environment tab, or the
# `envVars` list of the create-service call in README.md). A local run
# passes a filled-in copy with `docker run --env-file`.
# Anthropic
ANTHROPIC_API_KEY=
@ -17,11 +16,12 @@ AWS_BEARER_TOKEN_BEDROCK=
AWS_REGION_NAME=us-east-1
# Vertex AI (vertex_ai + vertex_ai_gpt columns).
# On the GCP VM, the default service-account ADC from the metadata server
# is used -- no JSON key file is needed. If you ever need to run outside
# GCP, also export GOOGLE_APPLICATION_CREDENTIALS=/path/to/sa.json.
# The service-account key JSON is the Render secret file
# `vertex-service-account.json`, mounted at /etc/secrets, and
# GOOGLE_APPLICATION_CREDENTIALS points google-auth at it.
VERTEXAI_PROJECT=
VERTEXAI_LOCATION=global
GOOGLE_APPLICATION_CREDENTIALS=/etc/secrets/vertex-service-account.json
# Azure AI Foundry (azure column — Claude models on Foundry)
AZURE_AI_API_KEY=
@ -35,19 +35,20 @@ AZURE_API_BASE=
AZURE_API_KEY=
# The publish PAT (mateo-berri, write access on BerriAI/litellm-docs)
# deliberately does NOT live in this file. Everything here lands in the
# process environment of pytest, the proxy, and the model-driven claude
# CLI, where any same-UID reader can lift it from /proc/<pid>/environ.
# Instead, install the token at /etc/litellm-compat-matrix-github-token
# (chmod 0600, single line); the service maps it in via systemd
# LoadCredential and run_daily.sh keeps it out of every child process
# env. Used to (a) resolve the latest stable release, (b) push the
# daily compat-matrix branch directly to BerriAI/litellm-docs, (c) open
# the same-repo PR, and (d) enable squash auto-merge on it. Scopes:
# deliberately is NOT an env var. Everything here lands in the process
# environment of pytest, the proxy, and the model-driven claude CLI,
# where any same-UID reader can lift it from /proc/<pid>/environ.
# Instead, the token is the Render secret file `github-token` (single
# line), mounted under CREDENTIALS_DIRECTORY, and run_daily.sh reads it
# from there and keeps it out of every child process env. Used to
# (a) resolve the latest stable release, (b) push the daily
# compat-matrix branch directly to BerriAI/litellm-docs, (c) open the
# same-repo PR, and (d) enable squash auto-merge on it. Scopes:
# classic `repo` + `workflow`, or fine-grained on BerriAI/litellm-docs
# with Contents:RW + Pull requests:RW + Workflows:RW.
# Manual runs export GITHUB_TOKEN instead, or skip publishing entirely
# with SKIP_PUBLISH=1 (only writes the matrix JSON locally).
CREDENTIALS_DIRECTORY=/etc/secrets
# Optional: the bedrock_mantle column is opt-in because the AWS account
# needs the Mantle (OpenAI-on-Bedrock) models enabled. Without this the
@ -59,9 +60,9 @@ AZURE_API_KEY=
# usually run them. Skipped cells are recorded as not_tested.
# COMPAT_OPENAI_GPT_CELLS=1
# Optional overrides; defaults are sensible for the cron VM.
# Optional overrides; defaults are sensible for the cron job.
# PROXY_PORT=4100
# LITELLM_WORKTREE=/home/mateo/litellm-cron-worktree
# LITELLM_WORKTREE=/home/populator/litellm-cron-worktree
# DOCS_REPO=BerriAI/litellm-docs
# DOCS_BRANCH=main
# DOCS_TARGET_PATH=src/data/compatibility-matrix.json

View file

@ -1,113 +0,0 @@
# systemd service for the Claude Code compatibility-matrix populator.
#
# Triggered by `litellm-compat-matrix.timer`; not started directly. The
# unit is a `Type=oneshot` so the timer's `OnCalendar=` semantics
# describe "run once per day" cleanly — there's no long-lived daemon to
# supervise; each invocation runs the populator end-to-end and exits.
#
# Install
# -------
#
# sudo cp tests/e2e/claude_code/cron_vm/litellm-compat-matrix.service /etc/systemd/system/
# sudo cp tests/e2e/claude_code/cron_vm/litellm-compat-matrix.timer /etc/systemd/system/
# sudo systemctl daemon-reload
# sudo systemctl enable --now litellm-compat-matrix.timer
#
# Paths are hard-coded to /home/mateo rather than using systemd's %h
# specifier. Why: in *system* units (this one), %h is expanded at
# parse time against the *manager's* home -- which is /root for PID 1
# -- and *not* against the User= directive. That mismatch makes
# ReadWritePaths point at /root/.cache (which doesn't exist), causing
# the namespace setup to fail with status=226/NAMESPACE before the
# script ever runs. The runtime user (`User=mateo`) must:
#
# * have a checkout of `BerriAI/litellm` at `~/litellm/litellm` so the
# publisher module is importable;
# * have a uv venv at `~/litellm/litellm/.venv` (created by
# `uv sync --frozen` inside that checkout once);
# * have `gh` already authenticated against an account with
# `pull-requests: write` on `BerriAI/litellm-docs`;
# * have provider credentials exported in `/etc/litellm-compat-matrix.env`
# (see `litellm-compat-matrix.env.example` in this directory);
# * have the mateo-berri publish PAT at
# `/etc/litellm-compat-matrix-github-token` (chmod 0600, single
# line), delivered via `LoadCredential=` below.
[Unit]
Description=Claude Code compatibility-matrix populator (oneshot)
Documentation=file:///home/mateo/litellm/litellm/tests/e2e/claude_code/cron_vm/README.md
Wants=network-online.target
After=network-online.target
[Service]
Type=oneshot
User=mateo
Group=mateo
# Provider credentials + any gh/PROXY_PORT overrides live here. Format
# is the standard `KEY=value` one line per env var.
EnvironmentFile=-/etc/litellm-compat-matrix.env
# The mateo-berri publish PAT is mapped in via the credential store, NOT
# the EnvironmentFile, so it never lands in the process environment that
# pytest, the proxy, and the model-driven claude CLI inherit (any
# same-UID process can read /proc/<pid>/environ). run_daily.sh reads
# ${CREDENTIALS_DIRECTORY}/github-token and hands it to gh per call.
# Unlike EnvironmentFile= above, this is deliberately NOT optional: a
# missing token file fails the unit at start instead of 30 minutes in.
LoadCredential=github-token:/etc/litellm-compat-matrix-github-token
# systemd starts with a minimal PATH (~/usr/local/bin:/usr/bin:/bin).
# `uv` and `claude` are installed under the runtime user's `~/.local/bin`
# so we have to prepend it explicitly; otherwise run_daily.sh fails at
# the up-front command-presence check.
Environment=PATH=/home/mateo/.local/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
# `HOME` is auto-set to /home/mateo when User=mateo is honored, but be
# explicit so anything that reads $HOME (e.g. uv's cache lookup, the
# claude CLI's per-session dir) sees the right value even if a future
# refactor flips DynamicUser= or PrivateUsers= on.
Environment=HOME=/home/mateo
WorkingDirectory=/home/mateo/litellm/litellm
ExecStart=/home/mateo/litellm/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh
# 90 minutes is generous: cold runs do `git clone` + `uv sync` of a new
# tag's lockfile, which can take a couple of minutes on a 2-vCPU VM,
# plus the full feature x provider grid of pytest cells hitting several
# cloud providers.
TimeoutStartSec=90min
# A failed run shouldn't restart automatically — the next timer fire is
# the right retry. Reruns of the same day's matrix are idempotent.
Restart=no
# Security hardening: the populator only reads the litellm checkout and
# the env-file; everything else it writes lives in either the worktree
# (managed) or `/tmp` (cleaned up by tempfile).
#
# ReadWritePaths whitelist:
# * litellm-cron-worktree - the long-lived stable-tag checkout +
# its `.venv` (`uv sync` rewrites every
# run) + `.uv-bin` (pinned `uv` binary
# cache).
# * .cache - uv's wheel cache (~/.cache/uv) so we
# don't redownload pinned deps each run.
# * .claude - `claude` CLI's per-session state under
# `~/.claude/projects/<sha>/`; created
# on every `claude --print` invocation.
# * .config/gh - `gh` CLI host config; technically not
# needed when we pass GH_TOKEN inline,
# but cheap to whitelist and prevents
# future regressions if a code path
# ever falls back to the host config.
# * /tmp - mktemp -d workdir + proxy logs.
NoNewPrivileges=true
ProtectSystem=strict
ProtectHome=read-only
ReadWritePaths=/home/mateo/litellm-cron-worktree /home/mateo/.cache /home/mateo/.claude /home/mateo/.config/gh /tmp
PrivateTmp=true
[Install]
WantedBy=multi-user.target

View file

@ -1,25 +0,0 @@
# Daily timer for the compatibility-matrix populator.
#
# 06:00 UTC matches the original GitHub Actions cron schedule; chosen so
# operators in US/EU timezones see fresh PRs at the start of their work
# day.
#
# `Persistent=true` causes a missed run (VM was off / suspended) to
# fire the next time the timer is started, which is the property we
# want for a once-a-day job: the matrix should refresh as soon as the
# VM is reachable again, not wait another 24h.
#
# `RandomizedDelaySec=10min` smears load if multiple matrix-style
# pipelines are ever colocated on the same VM in the future.
[Unit]
Description=Run the Claude Code compatibility-matrix populator daily
[Timer]
OnCalendar=*-*-* 06:00:00 UTC
Persistent=true
RandomizedDelaySec=10min
Unit=litellm-compat-matrix.service
[Install]
WantedBy=timers.target

View file

@ -1,8 +1,8 @@
#!/usr/bin/env bash
# Daily Claude Code compatibility-matrix populator.
#
# Runs from the GCP VM `litellm-compatibility-matrix-populator` via the
# systemd timer in this directory. The flow is:
# Runs daily as the Render cron job `litellm-compat-matrix`, built from
# the Dockerfile in this directory (see README.md). The flow is:
#
# 1. Resolve the latest LiteLLM final release tag from the GitHub
# Releases API.
@ -33,12 +33,12 @@
# rather than spawning a new one. If the JSON is byte-identical to the
# docs branch, we skip the push entirely.
#
# Required commands on $PATH: git, uv, gh, jq, curl, claude, npm.
# Required commands on $PATH: git, uv, gh, jq, curl, claude.
# Required state: a litellm checkout at $LITELLM_REPO (this file lives in
# it), $WORKTREE is created on first run, gh is already authenticated.
# it); $WORKTREE is created on first run.
#
# Override any default by setting the matching env var; see the systemd
# unit for the production wiring.
# Override any default by setting the matching env var; see README.md
# for the production wiring.
set -Eeuo pipefail
@ -52,12 +52,12 @@ DOCS_TARGET_PATH="${DOCS_TARGET_PATH:-src/data/compatibility-matrix.json}"
SKIP_PUBLISH="${SKIP_PUBLISH:-0}"
PYTEST_K="${PYTEST_K:-}"
# The e2e suite uses PEP 695 `type` aliases, so the venv needs Python
# >= 3.12 (also what repo CI runs) even when the VM's system python is
# >= 3.12 (also what repo CI runs) even when the host's system python is
# older. uv fetches a managed CPython of this version on first use --
# checksum-verified against the manifest baked into the pinned uv
# binary -- and installs it under ${WORKTREE}/.uv-python (see
# UV_PYTHON_INSTALL_DIR below) so it lives inside the one tree the
# systemd sandbox lets us write to.
# UV_PYTHON_INSTALL_DIR below) so everything the run writes lives inside
# the worktree.
CRON_PYTHON_VERSION="${CRON_PYTHON_VERSION:-3.12}"
# Merge method for auto-merge. BerriAI/litellm-docs only allows squash
# merges (merge-commit and rebase are disabled at the repo level), so
@ -113,9 +113,9 @@ for cmd in git uv gh jq curl claude; do
done
# Publishing pushes the branch straight to BerriAI/litellm-docs and opens
# the PR as mateo-berri, who has write access on the docs repo. Under
# systemd the PAT arrives as a file via LoadCredential=, NOT via the
# EnvironmentFile: several suite cells let the model-driven claude CLI
# the PR as mateo-berri, who has write access on the docs repo. On
# Render the PAT arrives as a secret file under ${CREDENTIALS_DIRECTORY},
# NOT via an env var: several suite cells let the model-driven claude CLI
# read arbitrary files as this user, and /proc/<pid>/environ of the
# script, pytest, and the proxy would hand an env-borne token to any
# same-UID reader. Kept as an unexported shell variable and passed per
@ -125,13 +125,18 @@ done
# quota.
if [[ -z "${GITHUB_TOKEN:-}" && -n "${CREDENTIALS_DIRECTORY:-}" && -f "${CREDENTIALS_DIRECTORY}/github-token" ]]; then
GITHUB_TOKEN="$(<"${CREDENTIALS_DIRECTORY}/github-token")"
log "publish token source: systemd credential store"
log "publish token source: ${CREDENTIALS_DIRECTORY}/github-token"
elif [[ -n "${GITHUB_TOKEN:-}" ]]; then
log "publish token source: process environment"
fi
if [[ "${SKIP_PUBLISH}" != "1" ]]; then
[[ -n "${GITHUB_TOKEN:-}" ]] \
|| die "publish token required: /etc/litellm-compat-matrix-github-token via LoadCredential under systemd, or an exported GITHUB_TOKEN for manual runs (or set SKIP_PUBLISH=1)"
|| die "publish token required: the github-token secret file under CREDENTIALS_DIRECTORY, or an exported GITHUB_TOKEN for manual runs (or set SKIP_PUBLISH=1)"
# The stale-PR sweep below closes only PRs this account opened, so the
# login is resolved from the token once rather than hardcoded.
PUBLISH_LOGIN="$(GH_TOKEN="${GITHUB_TOKEN}" gh api user --jq .login)" \
|| die "could not resolve the publishing account from the github token"
log "publishing as ${PUBLISH_LOGIN}"
fi
# ---------------------------------------------------------------------------
@ -205,7 +210,7 @@ log "local claude code: ${CLAUDE_CODE_VERSION}"
if [[ ! -d "${WORKTREE}/.git" ]]; then
log "first run: cloning litellm into ${WORKTREE}"
mkdir -p "$(dirname "${WORKTREE}")"
git clone https://github.com/BerriAI/litellm.git "${WORKTREE}"
git clone --filter=blob:none https://github.com/BerriAI/litellm.git "${WORKTREE}"
fi
log "updating worktree to ${LITELLM_VERSION}"
@ -221,40 +226,27 @@ git -C "${WORKTREE}" clean -fdx -e .venv -e .uv-bin -e .uv-python
git -C "${WORKTREE}" checkout --force "${LITELLM_VERSION}"
# Always rebuild tests/e2e/ in the worktree from the dev checkout,
# regardless of what the resolved ${LITELLM_VERSION} tag ships. Two
# reasons:
# regardless of what the resolved ${LITELLM_VERSION} tag ships: the
# matrix populator's job is to exercise *today's* tests against the
# latest stable proxy, and the dev checkout carries the most recent
# test fixes that haven't yet rolled into a stable release.
#
# * The matrix populator's job is to exercise *today's* tests against
# the latest stable proxy. The dev checkout carries the most recent
# test fixes that haven't yet rolled into a stable release, and we
# want every cron run to pick those up the moment they land on
# ${LITELLM_REPO}, not whenever the next stable release happens.
# * The tag's own tests/e2e/ ships the full EKS e2e harness, whose
# top-level conftest.py imports modules (e2e_db, lifecycle,
# otel_client, ...) that the stable venv does not install. Copying
# the whole tree would make pytest collection blow up on those
# imports.
#
# So the shim is a fresh `rm -rf` of tests/e2e/ followed by copying ONLY
# the claude_code suite plus the shared transport helpers it imports.
# The whole tree is copied rather than an allowlist of the helpers the
# suite imports: the helpers import each other (proxy_client ->
# e2e_config -> fixture_mode -> ...), so a new edge in that graph turned
# an allowlist into a ModuleNotFoundError at conftest load. The tree's
# top-level conftest.py pulls in the full EKS harness (e2e_db,
# lifecycle, ...), which the stable venv does not install, so the pytest
# run below points --confcutdir at claude_code/ and never loads it.
# pytest puts tests/e2e/ itself on sys.path (it has no __init__.py, while
# claude_code/ does), which is what resolves both the `claude_code.*`
# and the bare `proxy_client` / `e2e_http` imports inside the suite.
E2E_HELPER_FILES=(proxy_client.py e2e_http.py models.py e2e_config.py transport.py)
if [[ ! -d "${LITELLM_REPO}/tests/e2e/claude_code" ]]; then
die "no shim source at ${LITELLM_REPO}/tests/e2e/claude_code"
fi
for helper in "${E2E_HELPER_FILES[@]}"; do
[[ -f "${LITELLM_REPO}/tests/e2e/${helper}" ]] \
|| die "missing shim helper: ${LITELLM_REPO}/tests/e2e/${helper}"
done
log "shimming tests/e2e/claude_code/ + helpers from ${LITELLM_REPO} (always-overwrite)"
[[ -d "${LITELLM_REPO}/tests/e2e/claude_code" ]] \
|| die "no shim source at ${LITELLM_REPO}/tests/e2e/claude_code"
log "shimming tests/e2e/ from ${LITELLM_REPO} (always-overwrite)"
rm -rf "${WORKTREE}/tests/e2e"
mkdir -p "${WORKTREE}/tests/e2e"
cp -r "${LITELLM_REPO}/tests/e2e/claude_code" "${WORKTREE}/tests/e2e/"
for helper in "${E2E_HELPER_FILES[@]}"; do
cp "${LITELLM_REPO}/tests/e2e/${helper}" "${WORKTREE}/tests/e2e/"
done
cp -r "${LITELLM_REPO}/tests/e2e/." "${WORKTREE}/tests/e2e/"
# litellm pins an exact uv version in pyproject.toml's [tool.uv]
# `required-version` field, so a system uv that's newer or older
@ -305,10 +297,18 @@ fi
# actually serve. `--group proxy-dev` brings in pytest and the rest of
# what tests/e2e/claude_code/ needs. `--python` pins the venv to
# ${CRON_PYTHON_VERSION}; the first run after a version bump recreates
# the venv from scratch (a one-time cold sync).
# the venv from scratch (a one-time cold sync). `--no-install-project`
# leaves litellm itself out: the tag builds a Rust extension through
# maturin, which needs a C and Rust toolchain the image does not carry,
# so the published PyPI wheel (what users install) goes in right after,
# and every later `uv run` passes `--no-sync` so uv never tries to put
# the source build back.
export UV_PYTHON_INSTALL_DIR="${WORKTREE}/.uv-python"
log "uv sync --frozen --group proxy-dev --extra proxy --python ${CRON_PYTHON_VERSION} (uv ${PINNED_UV_VERSION:-system})"
(cd "${WORKTREE}" && "${WORKTREE_UV}" sync --frozen --group proxy-dev --extra proxy --python "${CRON_PYTHON_VERSION}")
log "uv sync --frozen --group proxy-dev --extra proxy --no-install-project --python ${CRON_PYTHON_VERSION} (uv ${PINNED_UV_VERSION:-system})"
(cd "${WORKTREE}" && "${WORKTREE_UV}" sync --frozen --group proxy-dev --extra proxy --no-install-project --python "${CRON_PYTHON_VERSION}")
LITELLM_WHEEL_VERSION="${LITELLM_VERSION#v}"
log "installing the published litellm==${LITELLM_WHEEL_VERSION} wheel from PyPI"
"${WORKTREE_UV}" pip install --python "${WORKTREE}/.venv/bin/python" --no-deps --no-build "litellm==${LITELLM_WHEEL_VERSION}"
PROXY_CONFIG="${WORKTREE}/tests/e2e/claude_code/test_config.yaml"
[[ -f "${PROXY_CONFIG}" ]] || die "proxy config not found at ${PROXY_CONFIG} (shim incomplete?)"
@ -321,10 +321,10 @@ log "starting proxy on 127.0.0.1:${PROXY_PORT}"
# Bind the proxy to loopback only. The populator proxy is talked to
# exclusively by the pytest run on the same host (the health check and
# the test env set `LITELLM_PROXY_URL=http://127.0.0.1:...`),
# so there's no reason to expose it on the VM's external interfaces.
# so there's no reason to expose it on the container's external interfaces.
# Without `--host`, `litellm` defaults to 0.0.0.0, which combined with
# the predictable default `LITELLM_MASTER_KEY=sk-cron-matrix` would
# allow anything that can reach :${PROXY_PORT} on the VM to authenticate
# allow anything that can reach :${PROXY_PORT} on the host to authenticate
# and burn upstream provider credentials.
#
# `setsid` puts the proxy in its own session+pgroup so cleanup() can
@ -334,7 +334,7 @@ log "starting proxy on 127.0.0.1:${PROXY_PORT}"
setsid env LITELLM_MASTER_KEY="${PROXY_API_KEY}" bash -c '
echo "$$" > "$0"
cd "$1"
exec "$2" run litellm --config "$3" --host 127.0.0.1 --port "$4"
exec "$2" run --no-sync litellm --config "$3" --host 127.0.0.1 --port "$4"
' "${PROXY_PID_FILE}" "${WORKTREE}" "${WORKTREE_UV}" "${PROXY_CONFIG}" "${PROXY_PORT}" \
>"${WORKDIR}/proxy.log" 2>&1 &
disown
@ -359,6 +359,7 @@ RESULTS_JSON="${WORKDIR}/compat-results.json"
# the cron skips them if/when they land in the suite.
PYTEST_ARGS=(
tests/e2e/claude_code/
--confcutdir=tests/e2e/claude_code
"--ignore-glob=*_unit_tests*"
)
if [[ -n "${PYTEST_K}" ]]; then
@ -373,7 +374,7 @@ set +e
&& LITELLM_PROXY_URL="http://127.0.0.1:${PROXY_PORT}" \
LITELLM_MASTER_KEY="${PROXY_API_KEY}" \
COMPAT_RESULTS_PATH="${RESULTS_JSON}" \
"${WORKTREE_UV}" run pytest "${PYTEST_ARGS[@]}"
"${WORKTREE_UV}" run --no-sync pytest "${PYTEST_ARGS[@]}"
)
PYTEST_EXIT=$?
set -e
@ -392,7 +393,7 @@ MATRIX_JSON="${WORKDIR}/compatibility-matrix.json"
log "building ${MATRIX_JSON}"
(
cd "${WORKTREE}" \
&& "${WORKTREE_UV}" run python "${POPULATOR_DIR}/build_matrix.py" \
&& "${WORKTREE_UV}" run --no-sync python "${POPULATOR_DIR}/build_matrix.py" \
--manifest "${WORKTREE}/tests/e2e/claude_code/manifest.yaml" \
--results "${RESULTS_JSON}" \
--output "${MATRIX_JSON}" \
@ -405,8 +406,9 @@ log "building ${MATRIX_JSON}"
# ---------------------------------------------------------------------------
if [[ "${SKIP_PUBLISH}" == "1" ]]; then
cp "${MATRIX_JSON}" "${LITELLM_REPO}/compatibility-matrix.json"
log "SKIP_PUBLISH=1; matrix written to ${LITELLM_REPO}/compatibility-matrix.json"
cp "${MATRIX_JSON}" "${HOME}/compatibility-matrix.json"
log "SKIP_PUBLISH=1; matrix saved to ${HOME}/compatibility-matrix.json and printed below"
cat "${MATRIX_JSON}"
exit 0
fi
@ -415,7 +417,7 @@ BRANCH_NAME="compat-matrix/${LITELLM_VERSION}-${CLAUDE_CODE_VERSION}-${DATE_UTC}
DOCS_CLONE="${WORKDIR}/litellm-docs"
log "cloning ${DOCS_REPO}@${DOCS_BRANCH}"
gh repo clone "${DOCS_REPO}" "${DOCS_CLONE}" -- --depth 1 --branch "${DOCS_BRANCH}"
GH_TOKEN="${GITHUB_TOKEN}" gh repo clone "${DOCS_REPO}" "${DOCS_CLONE}" -- --depth 1 --branch "${DOCS_BRANCH}"
cd "${DOCS_CLONE}"
git config user.email "litellm-bot@berri.ai"
@ -452,7 +454,7 @@ log "checking for green->red regressions vs the published matrix"
set +e
REGRESSION_REPORT="$(
cd "${WORKTREE}" \
&& "${WORKTREE_UV}" run python "${POPULATOR_DIR}/check_regressions.py" \
&& "${WORKTREE_UV}" run --no-sync python "${POPULATOR_DIR}/check_regressions.py" \
--old "${PUBLISHED_MATRIX}" \
--new "${MATRIX_JSON}"
)"
@ -492,7 +494,7 @@ git commit -m "${COMMIT_MSG}"
#
# Plain --force (not --force-with-lease) is acceptable here: the
# compat-matrix/* branch is bot-owned, only this script ever writes to
# it, and runs are serialized by the systemd timer. --force-with-lease
# it, and runs are serialized by the cron schedule. --force-with-lease
# would require a fetch to populate the remote-tracking ref before each
# push and adds no safety in this single-writer setup.
PUBLISH_PUSH_URL="https://x-access-token:${GITHUB_TOKEN}@github.com/${DOCS_REPO}.git"
@ -553,7 +555,7 @@ Generated by \`tests/e2e/claude_code/cron_vm/run_daily.sh\`. Close without mergi
EOF
)"
log "opening PR from ${BRANCH_NAME} -> ${DOCS_REPO}:${DOCS_BRANCH} (as mateo-berri)"
log "opening PR from ${BRANCH_NAME} -> ${DOCS_REPO}:${DOCS_BRANCH} (as ${PUBLISH_LOGIN})"
# GH_TOKEN is mateo-berri's write-scoped token, the same identity used
# for release-listing above. The branch lives on ${DOCS_REPO} itself, so
# --head is a bare branch name (a same-repo PR), not `OWNER:BRANCH`.
@ -644,15 +646,25 @@ fi
#
# Non-fatal: a sweep failure (rate limit, transient API error) leaves
# stale PRs for the next run to retry; it must not fail the pipeline.
#
# The docs repo carries a few hundred open PRs, so the list has to page
# past gh's default 30 (and the earlier 100, which never reached a
# week-old compat-matrix PR and left it open for good).
#
# `compat-matrix/` is only a naming convention, so the prefix alone does
# not make a PR this job's: a contributor can open a fork PR under that
# name. Only PRs the publishing account itself opened from a branch on
# the docs repo qualify; anything else stays untouched.
log "sweeping stale compat-matrix PRs (keeping ${BRANCH_NAME})"
set +e
STALE_PRS="$(
GH_TOKEN="${GITHUB_TOKEN}" gh pr list \
--repo "${DOCS_REPO}" \
--state open \
--limit 100 \
--json number,headRefName \
--jq '.[] | select(.headRefName | startswith("compat-matrix/")) | "\(.number)\t\(.headRefName)"'
--author "${PUBLISH_LOGIN}" \
--limit 1000 \
--json number,headRefName,isCrossRepository \
--jq '.[] | select((.headRefName | startswith("compat-matrix/")) and (.isCrossRepository | not)) | "\(.number)\t\(.headRefName)"'
)"
while IFS=$'\t' read -r stale_pr stale_head; do
[[ -z "${stale_pr}" ]] && continue
@ -660,7 +672,7 @@ while IFS=$'\t' read -r stale_pr stale_head; do
GH_TOKEN="${GITHUB_TOKEN}" gh pr close "${stale_pr}" \
--repo "${DOCS_REPO}" \
--delete-branch \
--comment "Superseded by the newer daily compat-matrix PR from \`${BRANCH_NAME}\`; the populator keeps only the most recent compat-matrix PR open." 2>&1 | sed 's/^/ /'
--comment "Superseded by the newer daily compat-matrix PR from \`${BRANCH_NAME}\`; the populator keeps only the most recent compat-matrix PR open" 2>&1 | sed 's/^/ /'
if [[ ${PIPESTATUS[0]} -eq 0 ]]; then
log "closed stale compat-matrix PR #${stale_pr} (${stale_head})"
else

View file

@ -345,6 +345,50 @@ def request_with_retry[T: RetryableResponse](
return issue()
PROVIDER_RATE_LIMIT_MARKER: Final = "litellm.RateLimitError"
PROVIDER_RATE_LIMIT_ATTEMPTS: Final = 4
PROVIDER_RATE_LIMIT_BACKOFF_SECONDS: Final = 5.0
def tolerate_provider_rate_limit[R: BaseModel](
issue: Callable[[], Result[R]],
*,
attempts: int = PROVIDER_RATE_LIMIT_ATTEMPTS,
sleep: Callable[[float], None] = time.sleep,
) -> Result[R]:
"""Retry a call up to `attempts` times while the proxy relays the provider's own 429;
any other outcome, the proxy's own 429 included, comes back at once."""
for attempt in range(1, attempts):
match issue():
case RateLimitedError(body=body, retry_after_seconds=retry_after) if PROVIDER_RATE_LIMIT_MARKER in body:
delay = retry_after or PROVIDER_RATE_LIMIT_BACKOFF_SECONDS * (1 << (attempt - 1))
print(
f"e2e-http: provider rate limit relayed by the proxy; retry {attempt}/{attempts - 1} in {delay}s",
flush=True,
)
sleep(delay)
case result:
return result
return issue()
class ProxyErrorDetail(BaseModel):
message: str
type: str
code: str
class _ProxyErrorBody(BaseModel):
error: ProxyErrorDetail
def relayed_provider_rate_limit(outcome: RateLimitedError) -> ProxyErrorDetail | None:
"""The provider's own 429 as the proxy relayed it, or None when the 429 is the proxy's own."""
if PROVIDER_RATE_LIMIT_MARKER not in outcome.body:
return None
return _ProxyErrorBody.model_validate_json(outcome.body).error
class ClassifiableResponse(Protocol):
"""What classifying an outcome reads off a response. requests.Response satisfies
it, and so does a fake, so the classification rules are testable on their own."""
@ -931,7 +975,10 @@ class PreparedForward:
def prepare_forward(
method: str, url: str, headers: dict[str, str], body: bytes | None,
method: str,
url: str,
headers: dict[str, str],
body: bytes | None,
) -> PreparedForward | NetworkError:
try:
with requests.Session() as session:
@ -950,7 +997,8 @@ def forward_prepared_stream(prepared: PreparedForward, timeout: float) -> Stream
except requests.RequestException as exc:
return NetworkError(message=str(exc))
return StreamHead(
resp.status_code, {name.lower(): value for name, value in resp.headers.items()},
resp.status_code,
{name.lower(): value for name, value in resp.headers.items()},
primed_steps(_stream_steps(resp)),
)

View file

@ -13,7 +13,13 @@ Each case asserts the feature actually happened, not just a 200. Coverage matrix
call, so a never-seen prefix must come back cached on its very first call
(Gemini's implicit caching cannot hit a cold prefix), the cached count must
cover the marked block, and the spend row must be billed below the uncached
price of the prompt.
price of the cached tokens. Vertex's cache create is nondeterministic:
identical bodies come back 200 or with the minimum-token 400 ("The cached
content is of 1 tokens"), in failure bursts of 45 seconds and more, so up to
eight never-seen prefixes are tried with a pause after each rejection. The
billing check prices the cached tokens rather than prompt_tokens, which
Vertex reports inclusive of the cached prefix on some calls and exclusive
of it on others.
- Anthropic (claude-haiku-4-5, direct): the same ``cache_control`` prefix over
the OpenAI-compatible route; the second call must report cache-read tokens > 0.
- OpenAI (gpt-5.6): automatic prompt caching needs no request marker, so the
@ -50,7 +56,8 @@ VERTEX_MODEL = "vertex_ai/gemini-2.5-flash"
ANTHROPIC_MODEL = "anthropic/claude-haiku-4-5-20251001"
OPENAI_MODEL = "openai/gpt-5.6"
VERTEX_CACHE_TTL: Final = "300s"
VERTEX_COLD_CALL_ATTEMPTS: Final = 3
VERTEX_COLD_CALL_ATTEMPTS: Final = 8
VERTEX_COLD_CALL_PAUSE_SECONDS: Final = 15.0
VERTEX_MINIMUM_CACHED_TOKENS: Final = 1024
CACHED_SHARE_OF_PROMPT: Final = 0.9
VERTEX_CACHE_REJECTION_MARKER: Final = "minimum token count to start explicit caching"
@ -156,26 +163,36 @@ def _cold_cache_call(send: Callable[[str], Result[ChatResponse]]) -> ChatRespons
return unwrap(result)
def _first_engaged_cold_call(send: Callable[[str], Result[ChatResponse]]) -> ChatResponse | None:
for attempt in range(1, VERTEX_COLD_CALL_ATTEMPTS + 1):
candidate = _cold_cache_call(send)
if candidate is not None and _cached_read_tokens(candidate.usage) >= VERTEX_MINIMUM_CACHED_TOKENS:
return candidate
if attempt < VERTEX_COLD_CALL_ATTEMPTS:
print(
f"cache_control: vertex did not engage the cache on cold attempt {attempt}/{VERTEX_COLD_CALL_ATTEMPTS}; "
f"pausing {VERTEX_COLD_CALL_PAUSE_SECONDS}s before the next never-seen prefix",
flush=True,
)
time.sleep(VERTEX_COLD_CALL_PAUSE_SECONDS)
return None
def _first_cold_call_reads_cache(model: str, send: Callable[[str], Result[ChatResponse]]) -> ChatResponse:
completion: Final = next(
(
candidate
for candidate in (_cold_cache_call(send) for _ in range(VERTEX_COLD_CALL_ATTEMPTS))
if candidate is not None and _cached_read_tokens(candidate.usage) >= VERTEX_MINIMUM_CACHED_TOKENS
),
None,
)
completion: Final = _first_engaged_cold_call(send)
assert completion is not None, (
f"{model}: {VERTEX_COLD_CALL_ATTEMPTS} never-seen prompts marked with cache_control were each either "
f"rejected by Vertex's minimum-token check or served with fewer than {VERTEX_MINIMUM_CACHED_TOKENS} "
"cached tokens on their first call; explicit context caching did not engage"
f"{model}: {VERTEX_COLD_CALL_ATTEMPTS} never-seen prompts marked with cache_control, spread over "
f"{VERTEX_COLD_CALL_PAUSE_SECONDS * (VERTEX_COLD_CALL_ATTEMPTS - 1):.0f}s, were each either rejected by "
f"Vertex's minimum-token check or served with fewer than {VERTEX_MINIMUM_CACHED_TOKENS} cached tokens on "
"their first call; explicit context caching did not engage"
)
assert completion.choices, f"{model}: cached call returned no choices: {completion}"
usage: Final = completion.usage
cached: Final = _cached_read_tokens(usage)
assert usage and usage.prompt_tokens and cached >= CACHED_SHARE_OF_PROMPT * usage.prompt_tokens, (
f"{model}: only {cached} of {usage.prompt_tokens if usage else None} prompt tokens were served from the "
"cache; the cache_control block was not cached whole"
assert usage and usage.prompt_tokens, f"{model}: cached completion carried no prompt_tokens: {usage}"
assert cached >= CACHED_SHARE_OF_PROMPT * usage.prompt_tokens, (
f"{model}: only {cached} of {usage.prompt_tokens} prompt tokens were served from the cache; the "
"cache_control block was not cached whole"
)
return completion
@ -196,10 +213,11 @@ def _assert_billed_below_uncached_prompt(client: PassthroughClient, model: str,
assert row.prompt_tokens == usage.prompt_tokens, (
f"{model}: spend row prompt_tokens {row.prompt_tokens} != response prompt_tokens {usage.prompt_tokens}"
)
uncached_prompt_cost: Final = usage.prompt_tokens * _input_rate(client, model)
assert row.spend is not None and row.spend < uncached_prompt_cost, (
f"{model}: spend {row.spend} is not below the uncached price of the prompt alone ({uncached_prompt_cost} for "
f"{usage.prompt_tokens} tokens); cache-read pricing was not applied"
cached: Final = _cached_read_tokens(usage)
uncached_read_cost: Final = cached * _input_rate(client, model)
assert row.spend is not None and row.spend < uncached_read_cost, (
f"{model}: spend {row.spend} is not below the uncached price of the {cached} tokens read from the cache "
f"({uncached_read_cost}); cache-read pricing was not applied"
)

View file

@ -110,10 +110,11 @@ def _vision_messages() -> list[ChatMessage]:
def _assert_describes_cat(response: ChatResponse) -> None:
assert response.choices, f"vision returned no choices: {response}"
message = response.choices[0].message
content = (message.content if message else None) or ""
choice = response.choices[0]
content = (choice.message.content if choice.message else None) or ""
assert any(term in content.lower() for term in ("cat", "feline", "kitten", "kitty")), (
f"vision response did not describe the image: {content[:200]}"
f"vision response did not describe the image: {content[:200]!r} "
f"(finish_reason={choice.finish_reason!r}, usage={response.usage})"
)
@ -421,7 +422,11 @@ class TestVertexChatCompletions:
model = self._register(client, resources, "e2e-vertex-vision")
key = resources.key()
response = unwrap(client.proxy.chat(key, ChatBody(model=model, messages=_vision_messages(), max_tokens=32)))
response = unwrap(
client.proxy.chat(
key, ChatBody(model=model, messages=_vision_messages(), max_tokens=32, reasoning_effort="none")
)
)
_assert_describes_cat(response)
@pytest.mark.covers(

View file

@ -11,16 +11,28 @@ well-formed OCR document comes back. Per the e2e hard-fail contract, a case
fails when no proxy answers and also fails once a request reaches it: the proxy
fetches each provider's referenced secrets, so a
missing credential surfaces as a live provider error rather than silent green.
The provider keys are shared with other pipelines, so a provider's rate limit can
hold across the bounded retries. The case then accepts the gateway's faithful relay
of that 429 (throttling_error, code 429) as its second expected outcome; any other
non-success still fails at once.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Protocol
from typing import Final, Protocol
import pytest
from e2e_config import unique_marker
from e2e_http import assert_client_error, unwrap
from e2e_http import (
PROVIDER_RATE_LIMIT_ATTEMPTS,
RateLimitedError,
Success,
assert_client_error,
relayed_provider_rate_limit,
tolerate_provider_rate_limit,
)
from lifecycle import ResourceManager
from models import LiteLLMParamsBody, OcrBody, OcrDocument, OcrResponse
from proxy_client import ProxyClient
@ -146,24 +158,36 @@ def _assert_ocr_document(response: OcrResponse) -> None:
assert response.pages[0].markdown is not None, "first page has no markdown"
def _assert_provider_rate_limit_relayed(model: str, outcome: RateLimitedError) -> None:
detail: Final = relayed_provider_rate_limit(outcome)
assert detail is not None, f"{model}: the 429 is the gateway's own, not the provider's: {outcome.body}"
assert (detail.type, detail.code) == ("throttling_error", "429"), f"{model}: provider 429 relayed as {detail!r}"
print(
f"{model}: the provider's rate limit held across {PROVIDER_RATE_LIMIT_ATTEMPTS} attempts; "
f"the gateway relayed it as {detail.type} {detail.code}",
flush=True,
)
class TestRustOcrGateway:
@pytest.mark.parametrize("case", RUST_OCR_CASES, ids=_CASE_IDS)
def test_rust_ocr_response(
self, proxy: ProxyClient, resources: ResourceManager, case: _OcrCase
) -> None:
def test_rust_ocr_response(self, proxy: ProxyClient, resources: ResourceManager, case: _OcrCase) -> None:
model = f"rust-ocr-{case.suffix}-{unique_marker()}"
model_id = proxy.create_model(model, case.provider.litellm_params())
resources.defer(lambda: proxy.delete_model(model_id))
key = resources.key()
response = unwrap(proxy.ocr(key, OcrBody(model=model, document=case.document)))
_assert_ocr_document(response)
match tolerate_provider_rate_limit(lambda: proxy.ocr(key, OcrBody(model=model, document=case.document))):
case Success(data=response):
_assert_ocr_document(response)
case RateLimitedError() as outcome:
_assert_provider_rate_limit_relayed(model, outcome)
case outcome:
pytest.fail(f"{model}: {outcome!r}")
@pytest.mark.skip(reason="stage red: product gap, /v1/ocr 500s (aocr TypeError) on missing document instead of 400")
@pytest.mark.covers("llm.ocr.openai.input_validation.nonstream.works")
def test_missing_document_returns_error(
self, proxy: ProxyClient, resources: ResourceManager
) -> None:
def test_missing_document_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
model = f"rust-ocr-val-{unique_marker()}"
model_id = proxy.create_model(model, MistralOcr().litellm_params())
resources.defer(lambda: proxy.delete_model(model_id))
@ -174,4 +198,3 @@ class TestRustOcrGateway:
json=_OptionalOcrBody(model=model),
)
assert_client_error(result, "ocr missing document")

View file

@ -15,9 +15,10 @@ reliability behavior.
from __future__ import annotations
from collections.abc import Sequence
from collections.abc import Mapping, Sequence
from typing import Final
from pydantic import ValidationError
from pydantic import BaseModel, ValidationError
from proxy_client import ProxyClient
from e2e_config import CHEAP_OPENAI_MODEL, PROXY_BASE_URL, unique_marker
@ -367,6 +368,26 @@ def model_id_of(resp: StreamingResponse) -> str | None:
return resp.headers.get("x-litellm-model-id")
class _AzurePromptFilterResult(BaseModel):
content_filter_results: Mapping[str, object] | None = None
class _AzureAnnotatedChatBody(BaseModel):
prompt_filter_results: Sequence[_AzurePromptFilterResult] | None = None
def azure_prompt_filter_skipped(resp: StreamingResponse) -> bool:
"""True when Azure's 200 recorded no prompt-filter verdict (every `content_filter_results`
empty), so the prompt has to be sent again."""
try:
annotated: Final = _AzureAnnotatedChatBody.model_validate_json(resp.body)
except ValidationError:
return False
if not annotated.prompt_filter_results:
return False
return all(not entry.content_filter_results for entry in annotated.prompt_filter_results)
def _parsed(resp: StreamingResponse) -> ChatResponse | None:
try:
return ChatResponse.model_validate_json(resp.body)

View file

@ -16,10 +16,16 @@ failure: the provider refuses the prompt itself, on length or on policy, and
reroute those, not `fallbacks`. The policy refusal is a real one, from an Azure
OpenAI content filter rejecting a jailbreak prompt, and a control call first
proves the refusal reaches the customer as a 400 when no reroute is configured.
Azure intermittently answers without running its prompt filter at all (the
body's prompt_filter_results carry no verdict), which is not a pass, so both
calls send the prompt again, a bounded number of times, until the filter ran.
"""
from __future__ import annotations
from collections.abc import Callable
from typing import Final
import pytest
from complexity_router_client import ComplexityRouterClient
@ -29,6 +35,7 @@ from lifecycle import ResourceManager
from models import RouterSettingsOverride
from reliability_support import (
CONTENT_POLICY_PROMPT,
azure_prompt_filter_skipped,
chat_override,
completion_tokens_of,
content_of,
@ -64,6 +71,28 @@ def _assert_served_by_fallback(resp: StreamingResponse) -> None:
assert int(attempted) >= 1, f"x-litellm-attempted-fallbacks should be >= 1, got {attempted!r}"
AZURE_FILTER_ATTEMPTS: Final = 3
def _chat_once_azure_runs_its_filter(send: Callable[[], StreamingResponse]) -> StreamingResponse:
for attempt in range(1, AZURE_FILTER_ATTEMPTS):
resp = send()
if not azure_prompt_filter_skipped(resp):
return resp
print(
"e2e: azure answered without running its prompt filter; sending the jailbreak prompt again "
f"({attempt}/{AZURE_FILTER_ATTEMPTS - 1})",
flush=True,
)
return send()
def _filter_verdict(resp: StreamingResponse) -> str:
if azure_prompt_filter_skipped(resp):
return "azure skipped its prompt filter on every attempt"
return "the filter ran and let the prompt through"
class TestReliabilityFallbacks:
@pytest.mark.covers("reliability.fallback.5xx.routes_to_fallback")
def test_5xx_routes_to_fallback(
@ -124,17 +153,21 @@ class TestReliabilityFallbacks:
model_id = create_content_filtered_deployment(client.proxy, primary)
resources.defer(lambda: client.proxy.delete_model(model_id))
refused = chat_override(client.proxy, scoped_key, primary, f"{CONTENT_POLICY_PROMPT} {unique_marker()}")
refused = _chat_once_azure_runs_its_filter(
lambda: chat_override(client.proxy, scoped_key, primary, f"{CONTENT_POLICY_PROMPT} {unique_marker()}")
)
assert refused.status_code == 400, (
f"the content filter should have refused the jailbreak prompt with a 400, got {refused.status_code}: "
f"{refused.body[:300]}"
f"the content filter should have refused the jailbreak prompt with a 400, got {refused.status_code} "
f"({_filter_verdict(refused)}): {refused.body[:300]}"
)
resp = chat_override(
client.proxy,
scoped_key,
primary,
f"{CONTENT_POLICY_PROMPT} {unique_marker()}",
override=RouterSettingsOverride(content_policy_fallbacks=[{primary: ["gpt-5.5"]}]),
resp = _chat_once_azure_runs_its_filter(
lambda: chat_override(
client.proxy,
scoped_key,
primary,
f"{CONTENT_POLICY_PROMPT} {unique_marker()}",
override=RouterSettingsOverride(content_policy_fallbacks=[{primary: ["gpt-5.5"]}]),
)
)
_assert_served_by_fallback(resp)

View file

@ -293,6 +293,12 @@
"tests/integration/spend/test_vertex_context_cache.py::test_gemini_context_cache_creation_is_billed_once_at_its_own_rate": [
"quota_management.spend_tracking.vertex_context_cache.creation_and_read_tokens_charged"
],
"tests/integration/spend/test_spend_calculate.py::test_live_preview_entry_charges_cached_tokens_at_the_fresh_rate[gemini-live-2.5-flash-preview-native-audio-09-2025]": [
"quota_management.spend_tracking.spend_calculate.live_preview_cached_tokens_cost_fresh_rate"
],
"tests/integration/spend/test_spend_calculate.py::test_live_preview_entry_charges_cached_tokens_at_the_fresh_rate[gemini/gemini-live-2.5-flash-preview-native-audio-09-2025]": [
"quota_management.spend_tracking.spend_calculate.live_preview_cached_tokens_cost_fresh_rate"
],
"tests/integration/management/test_partial_update_sequences.py::test_restricted_actor_cannot_detach_key_from_project": [
"mgmt.key.update.project_detach_denied_to_restricted_actor"
],

View file

@ -18,3 +18,51 @@ def test_spend_calculate_rejects_unpriced_model_with_400(gateway: Gateway) -> No
assert error["type"] == "invalid_request_error", response.text
assert error["param"] == "model", response.text
assert model in string_value(error["message"]), response.text
GEMINI_LIVE_PREVIEW_MODELS: Final = (
"gemini-live-2.5-flash-preview-native-audio-09-2025",
"gemini/gemini-live-2.5-flash-preview-native-audio-09-2025",
)
@pytest.mark.parametrize("model", GEMINI_LIVE_PREVIEW_MODELS)
@pytest.mark.covers("quota_management.spend_tracking.spend_calculate.live_preview_cached_tokens_cost_fresh_rate")
def test_live_preview_entry_charges_cached_tokens_at_the_fresh_rate(gateway: Gateway, model: str) -> None:
def cost_with_cached_tokens(cached_tokens: int) -> float:
response: Final = gateway.request(
"POST",
"/spend/calculate",
{
"completion_response": {
"id": "chatcmpl-live-preview",
"object": "chat.completion",
"created": 1677652288,
"model": model,
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "live preview answer"},
"finish_reason": "stop",
}
],
"usage": {
"prompt_tokens": 101_000,
"completion_tokens": 0,
"total_tokens": 101_000,
"prompt_tokens_details": {"cached_tokens": cached_tokens},
},
}
},
)
assert response.status_code == 200, response.text
cost: Final = object_value(JSON_OBJECT.validate_json(response.text))["cost"]
assert isinstance(cost, int | float)
return float(cost)
cached_cost: Final = cost_with_cached_tokens(100_000)
fresh_cost: Final = cost_with_cached_tokens(0)
assert fresh_cost > 0, fresh_cost
assert cached_cost == pytest.approx(fresh_cost), (
f"the entry publishes no cached rate, so 100k cached tokens must bill like fresh ones: {cached_cost} vs {fresh_cost}"
)

View file

@ -3331,7 +3331,7 @@ def test_gemini_fine_tuned_model_request_consistency():
Assert the same transformation is applied to Fine tuned gemini 2.0 flash and gemini 2.0 flash
- Request 1: Fine tuned: vertex_ai/gemini/ft-uuid
- Request 2: vertex_ai/gemini-2.0-flash-001
- Request 2: vertex_ai/gemini-2.5-flash
"""
litellm.set_verbose = True
load_vertex_ai_credentials()
@ -3403,7 +3403,7 @@ def test_gemini_fine_tuned_model_request_consistency():
with patch.object(client, "post", new=MagicMock()) as mock_post_2:
try:
response_2 = completion(
model="vertex_ai/gemini-2.0-flash-001",
model="vertex_ai/gemini-2.5-flash",
messages=messages,
tools=tools,
tool_choice="auto",

View file

@ -5,7 +5,7 @@ import litellm.cost_calculator
import asyncio
import time
from typing import Optional
from typing import Final, Optional
from unittest.mock import MagicMock, patch
import pytest
@ -21,6 +21,9 @@ import json
import httpx
from litellm.types.utils import PromptTokensDetails
from litellm.litellm_core_utils.litellm_logging import CustomLogger
from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import (
convert_to_model_response_object,
)
class CustomLoggingHandler(CustomLogger):
@ -328,35 +331,39 @@ def test_whisper_azure():
assert round(cost, 5) == round(expected_cost, 5)
def test_dalle_3_azure_cost_tracking():
litellm.set_verbose = True
# model = "azure/dall-e-3-test"
# response = litellm.image_generation(
# model=model,
# prompt="A cute baby sea otter",
# api_version="2023-12-01-preview",
# api_base=os.getenv("AZURE_SWEDEN_API_BASE"),
# api_key=os.getenv("AZURE_SWEDEN_API_KEY"),
# base_model="dall-e-3",
# )
# print(f"response: {response}")
response = litellm.ImageResponse(
created=1710265780,
data=[
{
"b64_json": None,
"revised_prompt": "A close-up image of an adorable baby sea otter. Its fur is thick and fluffy to provide buoyancy and insulation against the cold water. Its eyes are round, curious and full of life. It's lying on its back, floating effortlessly on the calm sea surface under the warm sun. Surrounding the otter are patches of colorful kelp drifting along the gentle waves, giving the scene a touch of vibrancy. The sea otter has its small paws folded on its chest, and it seems to be taking a break from its play.",
"url": "test-azure-blob-url-with-sas-token",
}
],
def test_gpt_image_2_azure_cost_tracking():
azure_image_generation_response: Final = {
"created": 1758585600,
"data": [{"b64_json": "iVBORw0KGgo=", "revised_prompt": None, "url": None}],
"output_format": "png",
"quality": "low",
"size": "1024x1024",
"usage": {
"input_tokens": 12,
"input_tokens_details": {"image_tokens": 0, "text_tokens": 12},
"output_tokens": 196,
"output_tokens_details": {"image_tokens": 196, "text_tokens": 0},
"total_tokens": 208,
},
}
response: Final = convert_to_model_response_object(
response_object=azure_image_generation_response,
model_response_object=litellm.ImageResponse(),
response_type="image_generation",
hidden_params={"model": "gpt-image-2", "custom_llm_provider": "azure"},
)
response.usage = {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}
response._hidden_params = {"model": "dall-e-3", "model_id": None}
print(f"response hidden params: {response._hidden_params}")
cost = litellm.completion_cost(
completion_response=response, call_type="image_generation"
cost: Final = litellm.completion_cost(
completion_response=response,
model="azure/my-gpt-image-2-deployment",
custom_llm_provider="azure",
base_model="gpt-image-2",
call_type="image_generation",
)
assert cost > 0
pricing: Final = litellm.model_cost["azure/gpt-image-2"]
expected_cost: Final = pricing["input_cost_per_token"] * 12 + pricing["output_cost_per_image_token"] * 196
assert round(cost, 8) == round(expected_cost, 8)
def test_replicate_llama3_cost_tracking():

View file

@ -139,3 +139,4 @@ bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b
bedrock/eu-west-1/qwen.qwen3-next-80b-a3b
bedrock/eu-west-2/qwen.qwen3-next-80b-a3b
bedrock/sa-east-1/qwen.qwen3-next-80b-a3b
bedrock/eu-west-2/nvidia.nemotron-super-3-120b

View file

@ -1,12 +1,11 @@
import json
import os
from typing import Final
from unittest.mock import MagicMock, patch
import httpx
import pytest
from typing import Final
from unittest.mock import MagicMock, patch
import litellm
from litellm import ModelResponse
from litellm.llms.bedrock.chat.converse_transformation import AmazonConverseConfig
@ -802,13 +801,13 @@ def test_output_config_format_translated_to_native_output_config_converse():
}
result = config._transform_request(
model="bedrock/converse/us.anthropic.claude-opus-4-7",
model="bedrock/converse/us.anthropic.claude-sonnet-4-6",
messages=[{"role": "user", "content": "hi"}],
optional_params={
"maxTokens": 256,
"thinking": {"type": "adaptive"},
"output_config": {
"effort": "xhigh",
"effort": "max",
"format": {"type": "json_schema", "schema": schema},
},
},
@ -817,7 +816,7 @@ def test_output_config_format_translated_to_native_output_config_converse():
)
additional = result.get("additionalModelRequestFields", {})
assert additional.get("output_config") == {"effort": "xhigh"}
assert additional.get("output_config") == {"effort": "max"}
assert "format" not in additional["output_config"]
assert result["outputConfig"]["textFormat"]["type"] == "json_schema"
parsed_schema = json.loads(
@ -4292,6 +4291,84 @@ def test_translate_response_format_native_output_config(monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", old_env)
BEDROCK_OPUS_4_7_AND_4_8_MODELS: Final = (
"anthropic.claude-opus-4-7",
"global.anthropic.claude-opus-4-7",
"us.anthropic.claude-opus-4-7",
"eu.anthropic.claude-opus-4-7",
"au.anthropic.claude-opus-4-7",
"jp.anthropic.claude-opus-4-7",
"anthropic.claude-opus-4-8",
"global.anthropic.claude-opus-4-8",
"us.anthropic.claude-opus-4-8",
"eu.anthropic.claude-opus-4-8",
"au.anthropic.claude-opus-4-8",
"jp.anthropic.claude-opus-4-8",
"us-gov.anthropic.claude-opus-4-8",
"us-gov-west-1/anthropic.claude-opus-4-8",
"us-gov-east-1/anthropic.claude-opus-4-8",
)
CAPITAL_RESPONSE_FORMAT: Final = {
"type": "json_schema",
"json_schema": {
"name": "capital",
"schema": {
"type": "object",
"properties": {"city": {"type": "string"}, "country": {"type": "string"}},
"required": ["city", "country"],
"additionalProperties": False,
},
},
}
def _converse_request_for_json_schema(model: str, stream: bool) -> tuple[dict, dict]:
config = AmazonConverseConfig()
optional_params = config.map_openai_params(
non_default_params={"response_format": CAPITAL_RESPONSE_FORMAT, "stream": stream},
optional_params={},
model=model,
drop_params=False,
)
request = config._transform_request(
model=model,
messages=[{"role": "user", "content": "Name the capital of France."}],
optional_params=optional_params,
litellm_params={},
headers={},
)
return optional_params, request
@pytest.mark.parametrize("model", BEDROCK_OPUS_4_7_AND_4_8_MODELS)
@pytest.mark.parametrize("stream", [False, True])
def test_opus_4_7_and_4_8_json_schema_sent_as_forced_tool_not_output_config(monkeypatch, model, stream):
"""Regression for issue #27846: Bedrock rejects outputConfig on Opus 4.7 and 4.8
(``output_config.format: Extra inputs are not permitted``), so json_schema has to
go out as the forced json_tool_call tool, streamed through fake_stream."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
optional_params, request = _converse_request_for_json_schema(model=model, stream=stream)
assert "outputConfig" not in request
assert [tool["toolSpec"]["name"] for tool in request["toolConfig"]["tools"]] == ["json_tool_call"]
assert request["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}}
assert optional_params.get("fake_stream", False) is stream
def test_sonnet_4_6_json_schema_still_uses_native_output_config(monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
optional_params, request = _converse_request_for_json_schema(model="us.anthropic.claude-sonnet-4-6", stream=True)
assert request["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["name"] == "capital"
assert "toolConfig" not in request
assert "fake_stream" not in optional_params
def test_translate_response_format_fallback_tool_call():
"""For unsupported models, should fall back to tool-call approach."""
config = AmazonConverseConfig()
@ -5417,9 +5494,14 @@ def test_cache_control_injection_tool_config_honors_ttl_for_regional_model_lacki
old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP")
old_cost = litellm.model_cost
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")
cost_map = dict(litellm.get_model_cost_map(url=""))
cost_map["jp.anthropic.claude-opus-4-7"] = {
k: v
for k, v in cost_map["jp.anthropic.claude-opus-4-7"].items()
if k != "cache_creation_input_token_cost_above_1hr"
}
litellm.model_cost = cost_map
try:
assert "cache_creation_input_token_cost_above_1hr" not in litellm.model_cost["jp.anthropic.claude-opus-4-7"]
assert "cache_creation_input_token_cost_above_1hr" in litellm.model_cost["anthropic.claude-opus-4-7"]
config = AmazonConverseConfig()
messages = [
@ -7500,3 +7582,92 @@ def test_eager_input_streaming_non_boolean_is_a_bad_request():
"us.anthropic.claude-sonnet-4-5-20250929-v1:0",
[_eager_openai_tool(eager_input_streaming="true")],
)
@pytest.mark.parametrize("model", ("anthropic.claude-opus-4-7", "us.anthropic.claude-opus-4-7"))
def test_converse_accepts_anthropic_default_temperature(model: str) -> None:
result: Final = litellm.utils.get_optional_params(
model=model,
custom_llm_provider="bedrock",
temperature=1,
drop_params=False,
)
assert result["temperature"] == 1
def test_get_supported_openai_params_drops_sampling_params_for_gpt5_models():
config = AmazonConverseConfig()
for model in [
"bedrock/converse/global.openai.gpt-5.6-luna",
"global.openai.gpt-5.6-luna",
"global.openai.gpt-5.6-sol",
"us.openai.gpt-5.6-terra",
"eu.openai.gpt-5.6-luna",
"openai.gpt-5.6-luna",
"bedrock/openai.gpt-5.6-luna",
]:
supported = config.get_supported_openai_params(model=model)
assert "temperature" not in supported
assert "top_p" not in supported
supported_oss = config.get_supported_openai_params(model="openai.gpt-oss-120b-1:0")
assert "temperature" in supported_oss
assert "top_p" in supported_oss
def test_map_openai_params_drops_temperature_and_top_p_when_drop_params_true():
config = AmazonConverseConfig()
for model in [
"bedrock/converse/global.openai.gpt-5.6-luna",
"openai.gpt-5.6-luna",
"eu.openai.gpt-5.6-luna",
]:
result = config.map_openai_params(
non_default_params={"temperature": 1.0, "top_p": 0.9, "max_tokens": 50},
optional_params={},
model=model,
drop_params=True,
)
assert "temperature" not in result
assert "topP" not in result
assert result.get("maxTokens") == 50
def test_map_openai_params_raises_unsupported_params_when_drop_params_false(monkeypatch):
monkeypatch.setattr(litellm, "drop_params", False)
config = AmazonConverseConfig()
for model in [
"bedrock/converse/global.openai.gpt-5.6-luna",
"openai.gpt-5.6-luna",
]:
with pytest.raises(litellm.utils.UnsupportedParamsError) as exc_info:
config.map_openai_params(
non_default_params={"temperature": 1.0},
optional_params={},
model=model,
drop_params=False,
)
assert "does not support temperature=1.0" in str(exc_info.value)
def test_map_openai_params_retains_sampling_params_for_supported_models():
config = AmazonConverseConfig()
result = config.map_openai_params(
non_default_params={"temperature": 0.7, "top_p": 0.8},
optional_params={},
model="openai.gpt-oss-120b-1:0",
drop_params=False,
)
assert result.get("temperature") == 0.7
assert result.get("topP") == 0.8
def test_supports_sampling_params_prefixed_and_anthropic_fallback(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setitem(
litellm.model_cost,
"global.custom-test-reasoning-model",
{"supports_sampling_params": False},
)
assert AmazonConverseConfig._supports_sampling_params("custom-test-reasoning-model") is False
assert AmazonConverseConfig._supports_sampling_params("anthropic.claude-custom-unregistered") is True

View file

@ -1,4 +1,10 @@
import base64
import binascii
import datetime
import json
import struct
from collections.abc import AsyncIterator, Mapping, Sequence
from typing import Final
from unittest.mock import AsyncMock, MagicMock
import httpx
@ -13,6 +19,7 @@ from litellm.llms.bedrock.chat.invoke_handler import (
make_sync_call,
)
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.types.utils import ModelResponseStream
def test_transform_thinking_blocks_with_redacted_content():
@ -704,3 +711,91 @@ async def test_async_invoke_streaming_non_200_forwards_bedrock_response_headers(
)
assert exc_info.value.response.headers["x-amzn-requestid"] == "req-non200-async"
def _bedrock_event_stream_frame(chunk: Mapping[str, object]) -> bytes:
def header(name: str, value: str) -> bytes:
return bytes([len(name)]) + name.encode() + bytes([7]) + struct.pack(">H", len(value)) + value.encode()
headers: Final = header(":event-type", "chunk") + header(":content-type", "application/json") + header(
":message-type", "event"
)
payload: Final = json.dumps({"bytes": base64.b64encode(json.dumps(chunk).encode()).decode()}).encode()
prelude: Final = struct.pack(">II", 12 + len(headers) + len(payload) + 4, len(headers))
body: Final = prelude + struct.pack(">I", binascii.crc32(prelude)) + headers + payload
return body + struct.pack(">I", binascii.crc32(body))
def _openai_stream_chunk(delta: Mapping[str, str], finish_reason: str | None = None) -> Mapping[str, object]:
return {
"id": "chatcmpl-1",
"object": "chat.completion.chunk",
"created": 1,
"model": "moonshot.kimi-k2-thinking",
"choices": [{"index": 0, "delta": delta, "finish_reason": finish_reason}],
}
_MOONSHOT_RAW_STREAM: Final = b"".join(
_bedrock_event_stream_frame(chunk)
for chunk in (
_openai_stream_chunk({"role": "assistant", "reasoning_content": "thinking"}),
_openai_stream_chunk({"content": '{"city": '}),
_openai_stream_chunk({"content": '"San Francisco"}'}),
_openai_stream_chunk({}, "stop"),
)
)
def _assert_moonshot_stream_content(chunks: Sequence[ModelResponseStream]) -> None:
assert "".join(chunk.choices[0].delta.content or "" for chunk in chunks) == '{"city": "San Francisco"}'
assert "".join(getattr(chunk.choices[0].delta, "reasoning_content", None) or "" for chunk in chunks) == "thinking"
assert [chunk.choices[0].finish_reason for chunk in chunks if chunk.choices[0].finish_reason] == ["stop"]
@pytest.fixture
def _aws_test_credentials(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIATEST")
monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "test-secret")
monkeypatch.setenv("AWS_REGION_NAME", "us-east-1")
@pytest.mark.parametrize("response_format", [None, {"type": "json_object"}])
def test_moonshot_invoke_stream_yields_openai_shaped_chunks(
_aws_test_credentials: None, response_format: Mapping[str, str] | None
) -> None:
raw_stream: Final = _MOONSHOT_RAW_STREAM
response: Final = MagicMock(status_code=200, headers={})
response.iter_bytes = lambda chunk_size=None: iter([raw_stream])
client: Final = HTTPHandler()
client.post = MagicMock(return_value=response)
stream: Final = litellm.completion(
model="bedrock/invoke/moonshot.kimi-k2-thinking",
messages=[{"role": "user", "content": "weather as json"}],
stream=True,
client=client,
**({"response_format": response_format} if response_format else {}),
)
_assert_moonshot_stream_content(list(stream))
@pytest.mark.asyncio
async def test_moonshot_invoke_async_stream_yields_openai_shaped_chunks(_aws_test_credentials: None) -> None:
async def _aiter_bytes(chunk_size: int | None = None) -> AsyncIterator[bytes]:
yield _MOONSHOT_RAW_STREAM
response: Final = MagicMock(status_code=200, headers={})
response.aiter_bytes = _aiter_bytes
client: Final = AsyncHTTPHandler()
client.post = AsyncMock(return_value=response)
stream: Final = await litellm.acompletion(
model="bedrock/invoke/moonshot.kimi-k2-thinking",
messages=[{"role": "user", "content": "weather as json"}],
stream=True,
response_format={"type": "json_object"},
client=client,
)
_assert_moonshot_stream_content([chunk async for chunk in stream])

View file

@ -1,12 +1,17 @@
import asyncio
import base64
import copy
import json
import os
import struct
import zlib
from datetime import datetime
from types import SimpleNamespace
from collections.abc import AsyncIterator, Mapping, Sequence
from typing import Final
from unittest.mock import Mock
import httpx
import pytest
# Ensure the project root is on the import path so `litellm` can be imported when
@ -3395,3 +3400,97 @@ def test_bedrock_invoke_eager_input_streaming_beta_not_duplicated_with_client_he
)
assert result["anthropic_beta"] == [FINE_GRAINED_TOOL_STREAMING_BETA]
def _bedrock_event_frame(payload: Mapping[str, object]) -> bytes:
def _header(name: str, value: str) -> bytes:
return (
bytes([len(name)])
+ name.encode()
+ bytes([7])
+ struct.pack(">H", len(value))
+ value.encode()
)
headers: Final = (
_header(":message-type", "event")
+ _header(":event-type", "chunk")
+ _header(":content-type", "application/json")
)
body: Final = json.dumps(
{"bytes": base64.b64encode(json.dumps(payload).encode()).decode()}
).encode()
prelude: Final = struct.pack(">II", 12 + len(headers) + len(body) + 4, len(headers))
prelude_crc: Final = struct.pack(">I", zlib.crc32(prelude))
message_crc: Final = struct.pack(">I", zlib.crc32(prelude + prelude_crc + headers + body))
return prelude + prelude_crc + headers + body + message_crc
class _GatedAsyncByteStream(httpx.AsyncByteStream):
def __init__(self, chunks: Sequence[bytes], gate: asyncio.Event) -> None:
self._chunks = chunks
self._gate = gate
async def __aiter__(self) -> AsyncIterator[bytes]:
yield self._chunks[0]
await self._gate.wait()
for chunk in self._chunks[1:]:
yield chunk
async def aclose(self) -> None:
return None
@pytest.mark.asyncio
async def test_get_async_streaming_response_iterator_yields_small_frame_before_upstream_pauses():
gate: Final = asyncio.Event()
response: Final = httpx.Response(
200,
stream=_GatedAsyncByteStream(
chunks=(
_bedrock_event_frame(
{
"type": "message_start",
"message": {
"id": "msg_test",
"type": "message",
"role": "assistant",
"content": [],
"model": "us.anthropic.claude-sonnet-4-6",
"usage": {"input_tokens": 3, "output_tokens": 1},
},
}
),
_bedrock_event_frame(
{
"type": "message_stop",
"usage": {"input_tokens": 3, "output_tokens": 9},
}
),
),
gate=gate,
),
)
iterator: Final = AmazonAnthropicClaudeMessagesConfig().get_async_streaming_response_iterator(
model="us.anthropic.claude-sonnet-4-6",
httpx_response=response,
request_body={"model": "us.anthropic.claude-sonnet-4-6"},
litellm_logging_obj=LiteLLMLoggingObj(
model="bedrock/us.anthropic.claude-sonnet-4-6",
messages=[{"role": "user", "content": "Hello"}],
stream=True,
call_type="chat",
start_time=datetime.now(),
litellm_call_id="test_small_frame_before_upstream_pauses",
function_id="test_small_frame_before_upstream_pauses",
),
)
first: Final = await asyncio.wait_for(anext(iterator), timeout=10)
assert first.startswith(b"event: message_start\n"), first
gate.set()
remaining: Final = tuple([chunk async for chunk in iterator])
assert any(chunk.startswith(b"event: message_stop\n") for chunk in remaining), remaining
await iterator.aclose()

View file

@ -2462,6 +2462,12 @@ def test_is_gemini_3_or_newer():
assert VertexGeminiConfig._is_gemini_3_or_newer("gemini-pro") == False
assert VertexGeminiConfig._is_gemini_3_or_newer("gemini-flash") == False
assert VertexGeminiConfig._is_gemini_3_or_newer("4965075652664360960") == False
assert VertexGeminiConfig._is_gemini_3_or_newer("gemini/4965075652664360960") == False
assert VertexGeminiConfig._is_gemini_3_or_newer("gemini/ft-uuid") == False
assert VertexGeminiConfig._is_gemini_3_or_newer("gemma-3-27b-it") == False
assert VertexGeminiConfig._is_gemini_3_or_newer("gemini/gemma-3-27b-it") == False
# Edge cases
assert VertexGeminiConfig._is_gemini_3_or_newer("") == False
@ -2505,6 +2511,26 @@ def test_gemini_3_reasoning_effort_maps_to_thinking_level(model: str):
assert "thinkingBudget" not in mapped["thinkingConfig"]
@pytest.mark.parametrize(
"model",
["4965075652664360960", "gemini/4965075652664360960", "gemini/ft-uuid", "gemma-3-27b-it"],
)
def test_fine_tuned_endpoint_and_gemma_get_no_gemini_3_default_temperature(model: str):
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
VertexGeminiConfig,
)
mapped = VertexGeminiConfig().map_openai_params(
non_default_params={"max_tokens": 10},
optional_params={},
model=model,
drop_params=False,
)
assert mapped["max_output_tokens"] == 10
assert "temperature" not in mapped
def _tool_call_messages(tool_call_id: str):

View file

@ -149,6 +149,49 @@ async def test_post_call_failure_hook_attributes_single_router_deployment(
)
@pytest.mark.asyncio
async def test_pre_routing_reject_spend_log_keeps_public_model_group(proxy_logging, make_user_api_key_auth, monkeypatch):
from litellm.proxy import proxy_server
from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload
recorded: list[dict] = []
class _RecordingLogger(CustomLogger):
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
recorded.append(kwargs)
monkeypatch.setattr(
proxy_server,
"llm_router",
litellm.Router(
model_list=[
{
"model_name": "internal-model",
"litellm_params": {"model": "openai/gpt-4.1", "api_key": "sk-test"},
}
]
),
)
monkeypatch.setattr(litellm, "callbacks", [_RecordingLogger()])
proxy_logging.alert_types = []
await proxy_logging.post_call_failure_hook(
request_data={"model": "internal-model", "messages": [{"role": "user", "content": "hi"}]},
original_exception=HTTPException(status_code=401, detail="blocked key"),
user_api_key_dict=make_user_api_key_auth(request_route="/chat/completions"),
route="/chat/completions",
)
assert len(recorded) == 1
assert recorded[0]["standard_logging_object"]["model_group"] == "internal-model"
now: Final = datetime.now()
payload = get_logging_payload(
kwargs={**recorded[0], "completion_start_time": now}, response_obj=None, start_time=now, end_time=now
)
assert payload["model"] == "openai/gpt-4.1"
assert payload["model_group"] == "internal-model"
@pytest.mark.asyncio
async def test_post_call_failure_hook_keeps_router_stamped_metadata_for_post_call_failures(
proxy_logging, make_user_api_key_auth, monkeypatch

View file

@ -11,6 +11,7 @@ from litellm.rust_bridge.catalog import (
CacheRule,
Context,
Delivery,
LoggerContext,
Route,
RouteContext,
RouteRule,
@ -89,6 +90,13 @@ def test_backend_rollouts_stay_on_python_when_global_rust_is_enabled(
assert catalog.decision(context) is Decision.PYTHON
def test_logger_rollout_obeys_the_global_switch() -> None:
assert catalog.rollout(LoggerContext()) is Rollout.RUST_OPT_IN
assert catalog.decision(LoggerContext()) is Decision.PYTHON
configuration.rust(True)
assert catalog.decision(LoggerContext()) is Decision.RUST_WITH_FALLBACK
def test_response_cache_rules_select_the_whole_backend_runtime() -> None:
rules: Final = (
CacheRule(Rollout.RUST_REQUIRED, backends=frozenset({"local"})),

View file

@ -0,0 +1,137 @@
import logging
from typing import Final
import pytest
import litellm
from litellm._logging import (
DiagnosticProcessingFilter,
_python_process_diagnostic,
redact_secrets,
session_id_var,
trace_id_var,
verbose_logger,
)
from litellm.constants import MINIMUM_CUSTOM_KEY_LENGTH
from litellm.litellm_core_utils.secret_redaction import (
_python_redact_internal_details,
_python_redact_string,
_python_redact_structured_value,
)
from litellm.rust_bridge import diagnostics, logger
def test_native_records_preserve_metadata_and_redact_before_custom_handlers(
caplog: pytest.LogCaptureFixture, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(verbose_logger, "handlers", [])
secret: Final = "sk-" + "a" * 48
message: Final = f"Authorization: Bearer {secret}"
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
logger.emit(
logging.WARNING, message, "native.rs", 42, "litellm_http", {"retry": True, "api_key": secret}, ("", "")
)
record: Final = caplog.records[0]
assert len(caplog.records) == 1
assert record.getMessage() == redact_secrets(message)
assert secret not in record.getMessage()
assert (record.pathname, record.lineno, record.funcName) == ("native.rs", 42, "litellm_http")
assert record.__dict__["rust_fields"]["retry"] is True
assert secret not in str(record.__dict__["rust_fields"])
def test_native_context_is_scoped_and_respects_correlation_setting(
caplog: pytest.LogCaptureFixture, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(litellm, "request_correlation_in_logs", True)
context_before: Final = logger.context()
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
logger.emit(logging.WARNING, "native", "native.rs", 1, "litellm_http", {}, ("session", "trace"))
monkeypatch.setattr(litellm, "request_correlation_in_logs", False)
logger.emit(logging.WARNING, "disabled", "native.rs", 2, "litellm_http", {}, ("hidden", "hidden"))
first, second = caplog.records
assert (first.__dict__["session_id"], first.__dict__["trace_id"]) == ("session", "trace")
assert "session_id" not in second.__dict__
assert "trace_id" not in second.__dict__
assert (session_id_var.get(), trace_id_var.get()) == context_before
def test_native_logging_observes_level_changes(caplog: pytest.LogCaptureFixture) -> None:
with caplog.at_level(logging.ERROR, logger="LiteLLM"):
assert not logger.enabled(logging.WARNING)
logger.emit(logging.WARNING, "filtered", "native.rs", 1, "litellm_http", {}, ("", ""))
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
assert logger.enabled(logging.WARNING)
logger.emit(logging.WARNING, "visible", "native.rs", 1, "litellm_http", {}, ("", ""))
assert [record.getMessage() for record in caplog.records] == ["visible"]
@pytest.mark.parametrize(
"text",
(
"Authorization: Bearer abcdefghijklmnop",
"s3_secret_access_key=secret123",
"postgres://user:pass@database.internal/name",
'{"type":"service_account","private_key":"secret123"}',
"GET /v1?key=abcdefghij&page=2",
),
)
def test_native_credential_patterns_match_python(text: str) -> None:
pytest.importorskip("litellm.rust_bridge._native")
from litellm.rust_bridge._native import NativeDiagnosticProcessor
processor: Final = NativeDiagnosticProcessor(MINIMUM_CUSTOM_KEY_LENGTH)
assert processor.redact_text(text) == _python_redact_string(text)
assert processor.redact_structured_text("api_key", "secret123") == _python_redact_structured_value(
"api_key", "secret123"
)
def test_native_client_redaction_matches_python() -> None:
pytest.importorskip("litellm.rust_bridge._native")
from litellm.rust_bridge._native import NativeDiagnosticProcessor
text: Final = "error at /etc/secrets/config on db.internal\nTraceback (most recent call last):\nsecret"
processor: Final = NativeDiagnosticProcessor(MINIMUM_CUSTOM_KEY_LENGTH)
assert processor.redact_client_message(text) == _python_redact_internal_details(text)
def test_native_diagnostic_batch_matches_python() -> None:
pytest.importorskip("litellm.rust_bridge._native")
from litellm.rust_bridge._native import NativeDiagnosticProcessor
message: Final = "é" * 110 + "sk-" + "q" * 48 + "界" * 1000
exception: Final = "document=" + "Q" * 200
stack: Final = "api_key=secret123"
leaves: Final = (("api_key", "secret123"), (None, "safe"))
processor: Final = NativeDiagnosticProcessor(MINIMUM_CUSTOM_KEY_LENGTH)
rust: Final = processor.process_diagnostic(message, exception, stack, leaves, (True, 20, 500))
python: Final = _python_process_diagnostic(message, exception, stack, leaves, True, 20, 500)
assert rust[:3] == python[:3]
assert tuple(rust[3]) == python[3]
assert rust[4] == python[4]
assert "sk-qq" not in rust[0]
assert len(rust[0]) <= 500
assert rust[3] == ["REDACTED", "safe"]
def test_missing_native_diagnostic_processor_falls_back_before_record_mutation(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setenv("LITELLM_RUST", "1")
diagnostics.PROCESSOR.override(None)
try:
record: Final = logging.makeLogRecord({"name": "LiteLLM", "levelno": logging.INFO, "msg": "api_key=secret123"})
assert DiagnosticProcessingFilter().filter(record) is True
assert record.getMessage() == "REDACTED"
finally:
diagnostics.PROCESSOR.reset()
def test_unsupported_unicode_uses_safe_python_redaction(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("LITELLM_RUST", "1")
assert redact_secrets("broken\ud800 api_key=secret123") == "broken\ud800 REDACTED"

View file

@ -1,4 +1,3 @@
import logging
from typing import Final
import httpx
@ -11,6 +10,7 @@ from litellm.rust_bridge import settings
from litellm.secret_managers.main import get_secret_str
from litellm.types.secret_managers.main import KeyManagementSettings, KeyManagementSystem
def test_url_policy_reads_the_litellm_globals(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(litellm, "user_url_validation", False)
monkeypatch.setattr(litellm, "user_url_allowed_hosts", ["docs.internal:8443"])
@ -57,13 +57,6 @@ def test_http_settings_ignores_environment_overrides(monkeypatch: pytest.MonkeyP
assert result.ssl_verify is True
def test_warn_reaches_the_litellm_logger(caplog: pytest.LogCaptureFixture) -> None:
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
settings.warn("ssl_ecdh_curve 'secp521r1' is not supported")
assert [record.getMessage() for record in caplog.records] == ["ssl_ecdh_curve 'secp521r1' is not supported"]
class _VaultSecrets(CustomSecretManager):
def __init__(self, secrets: dict[str, str]) -> None:
super().__init__(secret_manager_name="rust_bridge_settings_test")

View file

@ -38,6 +38,7 @@ from litellm.types.utils import (
Usage,
)
from litellm.types.videos.main import VideoObject
from litellm.utils import supports_prompt_caching
@pytest.fixture
@ -872,6 +873,40 @@ def test_default_image_cost_calculator(monkeypatch):
assert cost == 10485760
@pytest.mark.parametrize(
("model", "quality", "size", "priced_key", "pixels"),
[
("azure/dall-e-3", "standard", "1024x1024", "azure/standard/1024-x-1024/dall-e-3", 1024 * 1024),
("azure/dall-e-3", "hd", "1024x1792", "azure/hd/1024-x-1792/dall-e-3", 1024 * 1792),
("dall-e-3", "hd", "1024x1792", "azure/hd/1024-x-1792/dall-e-3", 1024 * 1792),
],
)
def test_default_image_cost_calculator_matches_provider_first_quality_key(
monkeypatch, model: str, quality: str, size: str, priced_key: str, pixels: int
):
from litellm.cost_calculator import default_image_cost_calculator
monkeypatch.setattr(
litellm,
"model_cost",
{
"azure/standard/1024-x-1024/dall-e-3": {"litellm_provider": "azure", "input_cost_per_pixel": 1e-08},
"azure/hd/1024-x-1792/dall-e-3": {"litellm_provider": "azure", "input_cost_per_pixel": 3e-08},
},
)
cost = default_image_cost_calculator(
model=model,
custom_llm_provider="azure",
quality=quality,
n=1,
size=size,
optional_params={},
)
assert cost == litellm.model_cost[priced_key]["input_cost_per_pixel"] * pixels
def test_cost_calculator_with_cache_creation():
from litellm import completion_cost
from litellm.types.utils import Choices, Message, Usage
@ -4502,3 +4537,99 @@ def test_cost_per_token_bedrock_nemotron_super_3_uses_eu_west_2_entry_not_us_rat
assert prompt_usd == pytest.approx(prompt_tokens * regional["input_cost_per_token"])
assert completion_usd == pytest.approx(completion_tokens * regional["output_cost_per_token"])
GPT_REALTIME_2_FAMILY: Final = (
"azure/gpt-realtime-2.1",
"azure/gpt-realtime-2.1-mini",
"gpt-realtime-2",
"gpt-realtime-2.1",
"gpt-realtime-2.1-mini",
)
def test_gpt_realtime_2_family_prices_audio_cache_writes_and_reads_alike(_local_model_cost_map: None) -> None:
audio_cache_rates: Final = {
model: (
litellm.model_cost[model].get("cache_read_input_audio_token_cost"),
litellm.model_cost[model].get("cache_creation_input_audio_token_cost"),
)
for model in GPT_REALTIME_2_FAMILY
}
# Azure publishes one cached-audio meter per gpt-realtime-2 deployment,
# https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/, checked 2026-09-23
assert all(read is not None and write == read for read, write in audio_cache_rates.values()), audio_cache_rates
assert len(audio_cache_rates) == len(GPT_REALTIME_2_FAMILY)
GEMINI_LIVE_NATIVE_AUDIO_CASES: Final = (
("gemini-live-2.5-flash-native-audio", "vertex_ai"),
("gemini-live-2.5-flash-preview-native-audio-09-2025", "vertex_ai"),
("gemini/gemini-live-2.5-flash-preview-native-audio-09-2025", "gemini"),
)
@pytest.mark.parametrize(("model", "provider"), GEMINI_LIVE_NATIVE_AUDIO_CASES)
def test_gemini_live_native_audio_carries_no_cached_input_rate(
_local_model_cost_map: None, model: str, provider: str
) -> None:
# the Vertex pricing table prints N/A for cached input on every Live row,
# https://cloud.google.com/vertex-ai/generative-ai/pricing, checked 2026-09-23
assert litellm.get_model_info(model, custom_llm_provider=provider)["cache_read_input_token_cost"] is None
prompt_usd, _ = cost_per_token(
model=model,
prompt_tokens=101_000,
completion_tokens=0,
custom_llm_provider=provider,
usage_object=Usage(
prompt_tokens=101_000,
completion_tokens=0,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=100_000),
),
)
fresh_usd, _ = cost_per_token(
model=model,
prompt_tokens=101_000,
completion_tokens=0,
custom_llm_provider=provider,
usage_object=Usage(prompt_tokens=101_000, completion_tokens=0),
)
assert prompt_usd == pytest.approx(fresh_usd), (
"with no cached rate the cached tokens bill at the input rate, so a phantom discount cannot appear"
)
assert prompt_usd > 0
@pytest.mark.parametrize(("model", "provider"), GEMINI_LIVE_NATIVE_AUDIO_CASES)
def test_gemini_live_native_audio_declares_prompt_caching_unsupported(
_local_model_cost_map: None, model: str, provider: str
) -> None:
# the Vertex context-caching supported-model lists contain no Live model while 2.5 Flash is listed,
# https://cloud.google.com/vertex-ai/generative-ai/docs/context-cache/context-cache-overview, checked 2026-09-23
assert litellm.get_model_info(model, custom_llm_provider=provider)["supports_prompt_caching"] is False
assert supports_prompt_caching(model=model, custom_llm_provider=provider) is False
assert supports_prompt_caching(model="gemini-2.5-flash", custom_llm_provider="vertex_ai") is True, (
"control: the helper swallows a lookup error into False, so without this a broken lookup reads as a pass"
)
@pytest.mark.parametrize(
"model",
["gemini-live-2.5-flash-native-audio", "vertex_ai/gemini-live-2.5-flash-native-audio"],
)
def test_gemini_live_native_audio_limits_and_capabilities_match_vendor_model_card(
_local_model_cost_map: None, model: str
) -> None:
info = litellm.get_model_info(model)
# the Vertex model card for gemini-live-2.5-flash-native-audio publishes these limits and flags,
# https://cloud.google.com/vertex-ai/generative-ai/docs/models, checked 2026-09-23
assert info["max_input_tokens"] == 131072
assert info["max_output_tokens"] == 65536
assert info["max_tokens"] == 65536
assert info["supports_response_schema"] is False
assert info["supports_url_context"] is False
assert info["supports_pdf_input"] is False

View file

@ -19,6 +19,16 @@ from litellm._logging import (
_COLOR_LOG_FORMAT,
_MAX_SCRUBBED_ACCESS_ARG,
_PLAIN_LOG_FORMAT,
ALL_LOGGERS,
AccessLogPathFilter,
AccessLogRedactionFilter,
CorrelationContextFilter,
CorrelationPlainFormatter,
DiagnosticProcessingFilter,
JsonFormatter,
LevelRoutingStreamHandler,
SecretRedactionFilter,
StdoutLogTruncationFilter,
_get_uvicorn_json_log_config,
_initialize_loggers_with_handler,
_parse_json_logs_env,
@ -33,15 +43,6 @@ from litellm._logging import (
verbose_logger,
verbose_proxy_logger,
verbose_router_logger,
ALL_LOGGERS,
AccessLogPathFilter,
AccessLogRedactionFilter,
CorrelationContextFilter,
CorrelationPlainFormatter,
JsonFormatter,
LevelRoutingStreamHandler,
SecretRedactionFilter,
StdoutLogTruncationFilter,
)
from litellm.constants import LITELLM_TRUNCATED_PAYLOAD_FIELD
from litellm.integrations.custom_logger import CustomLogger
@ -824,6 +825,75 @@ def test_secret_filter_keeps_truncated_traceback(monkeypatch):
assert "sk-1234567890abcdefghij" not in record.exc_text
@pytest.mark.parametrize("native", (False, True), ids=("python", "rust"))
def test_diagnostic_redaction_precedes_a_credential_cut(monkeypatch, native):
if native:
pytest.importorskip("litellm.rust_bridge._native")
monkeypatch.setenv("LITELLM_RUST", "1" if native else "0")
monkeypatch.setenv("MAX_STRING_LENGTH_STDOUT_LOG", "500")
monkeypatch.setenv("MAX_BASE64_LENGTH_STDOUT_LOG", "0")
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
secret = "sk-" + "q" * 48
record = _make_record(logging.INFO, "%s", ("é" * 110 + secret + "界" * 1000,))
assert DiagnosticProcessingFilter().filter(record) is True
assert len(record.getMessage()) <= 500
assert "sk-qq" not in record.getMessage()
def test_correlation_id_redacts_before_its_length_bound(monkeypatch):
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
secret = "sk-" + "q" * 48
token = set_trace_id("x" * 250 + secret)
try:
assert "sk-qq" not in trace_id_var.get()
assert len(trace_id_var.get()) <= 256
finally:
trace_id_var.reset(token)
@pytest.mark.parametrize("native", (False, True), ids=("python", "rust"))
def test_malformed_interpolation_still_scrubs_a_record(monkeypatch, native):
if native:
pytest.importorskip("litellm.rust_bridge._native")
monkeypatch.setenv("LITELLM_RUST", "1" if native else "0")
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
record = _make_record(logging.WARNING, "bad % api_key=secret123", ("value",))
record.color_message = "bad % api_key=secret123"
assert DiagnosticProcessingFilter().filter(record) is True
assert record.getMessage() == "REDACTED"
assert record.color_message == "REDACTED"
@pytest.mark.parametrize("native", (False, True), ids=("python", "rust"))
def test_key_pattern_template_keeps_the_rendered_redacted_line(monkeypatch, native):
if native:
pytest.importorskip("litellm.rust_bridge._native")
monkeypatch.setenv("LITELLM_RUST", "1" if native else "0")
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
record = _make_record(logging.INFO, "password=%s ok", ("hunter2",))
record.color_message = "password=%s ok"
assert DiagnosticProcessingFilter().filter(record) is True
assert record.getMessage() == "REDACTED ok"
assert record.color_message == "REDACTED ok"
def test_disabled_diagnostic_call_does_not_render_arguments(caplog):
class Unrenderable:
def __str__(self):
raise AssertionError("disabled call rendered its argument")
with caplog.at_level(logging.ERROR, logger="LiteLLM"):
verbose_logger.debug("hidden %s", Unrenderable())
assert not caplog.records
def test_truncation_filter_survives_json_reconfiguration():
"""The cap lives on the loggers, so swapping handlers (JSON mode) can't drop it."""
_turn_on_json()
@ -983,10 +1053,10 @@ _REQUEST_DUMP = "{'model': 'gpt-4', 'messages': [{'role': 'user', 'content': 'he
(CorrelationPlainFormatter(_PLAIN_LOG_FORMAT), JsonFormatter()),
ids=("plain", "json"),
)
def test_scrubbed_record_is_scanned_for_secrets_once(monkeypatch, formatter):
"""Every pass of the secret regex over a multi-megabyte debug line costs seconds of
event-loop time, so a formatter must not rescan what SecretRedactionFilter scrubbed."""
def test_scrubbed_record_scans_the_large_rendered_value_once(monkeypatch, formatter):
"""The raw format template gets its own check, while the large rendered value gets one scan."""
counting = _CountingPattern(secret_redaction._SECRET_RE)
monkeypatch.setenv("LITELLM_RUST", "0")
monkeypatch.setattr(secret_redaction, "_SECRET_RE", counting)
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
record = _make_record(logging.DEBUG, "receiving data: %s", (_REQUEST_DUMP,))
@ -997,14 +1067,15 @@ def test_scrubbed_record_is_scanned_for_secrets_once(monkeypatch, formatter):
assert _REQUEST_DUMP in rendered
assert "litellm_redacted" not in rendered
assert counting.calls == 1
assert counting.scanned_chars == len(f"receiving data: {_REQUEST_DUMP}")
assert counting.calls == 2
assert counting.scanned_chars == len(f"receiving data: {_REQUEST_DUMP}") + len("receiving data: %s")
def test_stamped_record_is_not_scanned_again(monkeypatch):
"""JSON mode puts the filter on a third-party logger and again on the root handler its
records propagate to, so the second filter must trust the stamp instead of rescanning."""
counting = _CountingPattern(secret_redaction._SECRET_RE)
monkeypatch.setenv("LITELLM_RUST", "0")
monkeypatch.setattr(secret_redaction, "_SECRET_RE", counting)
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
record = _make_record(logging.DEBUG, "receiving data: %s", (_REQUEST_DUMP,))
@ -1012,13 +1083,14 @@ def test_stamped_record_is_not_scanned_again(monkeypatch):
assert SecretRedactionFilter().filter(record) is True
assert SecretRedactionFilter().filter(record) is True
assert counting.calls == 1
assert counting.calls == 2
def test_caller_supplied_stamp_never_skips_the_scrub(monkeypatch):
"""The stamp is a private sentinel, so a caller passing extra={"litellm_redacted": True}
still gets the full scrub, and only the filter's own stamp lets a later pass skip it."""
counting = _CountingPattern(secret_redaction._SECRET_RE)
monkeypatch.setenv("LITELLM_RUST", "0")
monkeypatch.setattr(secret_redaction, "_SECRET_RE", counting)
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
record = _make_record(logging.DEBUG, "api_key=sk-1234567890abcdefghij")
@ -1619,3 +1691,75 @@ def test_access_log_path_filter_keeps_a_record_without_a_string_path_arg(monkeyp
exc_info=None,
)
assert AccessLogPathFilter().filter(record) is True
@pytest.mark.parametrize("native", (False, True), ids=("python", "rust"))
def test_diagnostic_filter_scrubs_exc_stack_and_nested_extras(monkeypatch, native):
if native:
pytest.importorskip("litellm.rust_bridge._native")
monkeypatch.setenv("LITELLM_RUST", "1" if native else "0")
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
secret = "sk-" + "q" * 48
try:
raise ValueError(f"upstream rejected {secret}")
except ValueError:
record = _make_record(logging.ERROR, "call failed", exc_info=sys.exc_info())
record.stack_info = f"Stack (most recent call last): {secret}"
record.payload = {
"api_key": secret,
"items": [secret, "ok"],
"tags": {secret},
"pair": (secret, "ok"),
"count": 2,
}
assert DiagnosticProcessingFilter().filter(record) is True
assert secret not in (record.exc_text or "")
assert secret not in (record.stack_info or "")
assert secret not in repr(record.payload)
assert record.payload["count"] == 2
@pytest.mark.parametrize("native", (False, True), ids=("python", "rust"))
def test_diagnostic_filter_stamps_records_so_a_second_pass_is_free(monkeypatch, native):
if native:
pytest.importorskip("litellm.rust_bridge._native")
monkeypatch.setenv("LITELLM_RUST", "1" if native else "0")
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
record = _make_record(logging.WARNING, "api_key=secret123")
diagnostic_filter = DiagnosticProcessingFilter()
assert diagnostic_filter.filter(record) is True
assert diagnostic_filter.filter(record) is True
assert record.getMessage() == "REDACTED"
@pytest.mark.parametrize("native", (False, True), ids=("python", "rust"))
def test_json_formatter_scrubs_unfiltered_extras(monkeypatch, native):
if native:
pytest.importorskip("litellm.rust_bridge._native")
monkeypatch.setenv("LITELLM_RUST", "1" if native else "0")
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
secret = "sk-" + "q" * 48
record = _make_record(logging.INFO, "response complete")
record.payload = {"api_key": secret, "nested": {"list": [secret]}}
rendered = JsonFormatter().format(record)
assert secret not in rendered
assert "REDACTED" in rendered
@pytest.mark.parametrize("native", (False, True), ids=("python", "rust"))
def test_diagnostic_filter_redacts_a_non_string_message_object(monkeypatch, native):
if native:
pytest.importorskip("litellm.rust_bridge._native")
monkeypatch.setenv("LITELLM_RUST", "1" if native else "0")
monkeypatch.setattr("litellm._logging._ENABLE_SECRET_REDACTION", True)
secret = "sk-" + "q" * 48
record = _make_record(logging.ERROR, {"api_key": secret})
assert DiagnosticProcessingFilter().filter(record) is True
assert secret not in record.getMessage()

View file

@ -19,7 +19,11 @@ from litellm._logging import (
verbose_proxy_logger,
verbose_router_logger,
)
from litellm.litellm_core_utils.secret_redaction import redact_internal_details, redact_string
from litellm.litellm_core_utils.secret_redaction import (
redact_internal_details,
redact_string,
redact_structured_value,
)
SECRET = "sk-proj-abc123def456ghi789jklmnopqrst"
@ -71,6 +75,17 @@ def test_redact_string_catches_secret_patterns():
assert redact_string(normal) == normal
@pytest.mark.parametrize("native", (False, True), ids=("python", "rust"))
def test_diagnostic_redaction_policy_matches_across_backends(monkeypatch: pytest.MonkeyPatch, native: bool) -> None:
if native:
pytest.importorskip("litellm.rust_bridge._native")
monkeypatch.setenv("LITELLM_RUST", "1" if native else "0")
assert redact_string("GET /v1?api_key=abcdefgh12345&page=2") == "GET /v1?REDACTED&page=2"
assert redact_structured_value("db_url", "postgresql://reader@example.org/database") == "REDACTED"
assert redact_internal_details("failed at /etc/service/keys on db.internal") == "failed at REDACTED on REDACTED"
@pytest.mark.parametrize(
"connection_string",
[

View file

@ -8,7 +8,7 @@ import logging
import os
import queue
import threading
from collections.abc import Callable, Iterator
from collections.abc import Callable, Iterator, Mapping
from concurrent.futures import Future, ThreadPoolExecutor
from datetime import datetime, timedelta, timezone
from pathlib import PurePath
@ -5283,23 +5283,31 @@ async def test_wrapper_async_does_not_fire_failure_hook_for_post_success_error(
) -> None:
"""Regression: an error raised after the deployment call already succeeded (e.g. inside
async_post_call_success_deployment_hook or post_call_processing) is not a deployment
attempt failure and must not reach async_post_call_failure_deployment_hook."""
attempt failure and must not reach async_post_call_failure_deployment_hook. The raising
callback is a guardrail because a plain logger's success hook error is isolated and
logged instead of propagating out of the call."""
class ExplodingSuccessLogger(CustomLogger):
class ExplodingSuccessGuardrail(CustomGuardrail):
def __init__(self) -> None:
super().__init__()
self.failure_calls: list[Exception] = []
super().__init__(guardrail_name="exploding")
self.failure_calls: tuple[Exception, ...] = ()
async def async_post_call_success_deployment_hook(self, request_data, response, call_type):
async def async_post_call_success_deployment_hook(
self, request_data: Mapping[str, object], response: LLMResponseTypes, call_type: CallTypes | None
) -> LLMResponseTypes | None:
raise RuntimeError("boom in success hook, model call itself succeeded")
async def async_post_call_failure_deployment_hook(
self, request_data, exception, call_type, fallback_depth=None
):
self.failure_calls.append(exception)
self,
request_data: Mapping[str, object],
exception: Exception,
call_type: CallTypes | None,
fallback_depth: int | None = None,
) -> None:
self.failure_calls = (*self.failure_calls, exception)
exploding_logger = ExplodingSuccessLogger()
monkeypatch.setattr(litellm, "callbacks", [exploding_logger])
exploding_guardrail: Final = ExplodingSuccessGuardrail()
monkeypatch.setattr(litellm, "callbacks", [exploding_guardrail])
with pytest.raises(RuntimeError, match="boom in success hook"):
await litellm.acompletion(
@ -5308,7 +5316,7 @@ async def test_wrapper_async_does_not_fire_failure_hook_for_post_success_error(
mock_response="this call succeeds",
)
assert exploding_logger.failure_calls == []
assert exploding_guardrail.failure_calls == ()
@pytest.mark.asyncio

View file

@ -541,26 +541,48 @@ def test_output_config_format_converted_for_bedrock_chat_invoke_request():
assert json.loads(last_content[-1]["text"]) == schema
def test_output_config_format_forwarded_for_bedrock_chat_invoke_request():
@pytest.mark.parametrize("model", ["anthropic.claude-opus-4-7", "us.anthropic.claude-opus-4-8"])
def test_output_config_format_inlined_for_bedrock_chat_invoke_opus_4_7_and_4_8(local_model_cost_map, model):
"""Bedrock rejects ``output_config.format`` on Claude Opus 4.7 and 4.8, so the
Invoke chat path inlines the schema into the last user message and keeps effort,
driven by the cost map alone (no capability stub)."""
schema = {"type": "object", "properties": {"answer": {"type": "string"}}}
result = AmazonAnthropicClaudeConfig().transform_request(
model=model,
messages=[{"role": "user", "content": "test"}],
optional_params={
"max_tokens": 100,
"output_config": {"effort": "xhigh", "format": {"type": "json_schema", "schema": schema}},
},
litellm_params={},
headers={},
)
assert result.get("output_config") == {"effort": "xhigh"}
assert json.loads(result["messages"][-1]["content"][-1]["text"]) == schema
def test_output_config_format_forwarded_for_bedrock_chat_invoke_request(local_model_cost_map):
"""Bedrock Invoke chat path forwards ``output_config.format`` alongside effort
for models with native structured-output support (Claude Opus 4.7)."""
for models with native structured-output support (Claude Sonnet 4.6)."""
schema_format = {
"type": "json_schema",
"schema": {"type": "object", "properties": {"answer": {"type": "string"}}},
}
result = AmazonAnthropicClaudeConfig().transform_request(
model="anthropic.claude-opus-4-7",
model="us.anthropic.claude-sonnet-4-6",
messages=[{"role": "user", "content": "test"}],
optional_params={
"max_tokens": 100,
"output_config": {"effort": "xhigh", "format": schema_format},
"output_config": {"effort": "max", "format": schema_format},
},
litellm_params={},
headers={},
)
assert result.get("output_config") == {"effort": "xhigh", "format": schema_format}
assert result.get("output_config") == {"effort": "max", "format": schema_format}
assert "answer" not in json.dumps(result["messages"])

6
uv.lock generated
View file

@ -4499,7 +4499,7 @@ wheels = [
[[package]]
name = "litellm"
version = "1.103.0"
version = "1.104.0"
source = { editable = "." }
dependencies = [
{ name = "aiohttp" },
@ -4959,12 +4959,12 @@ proxy-dev = [
[[package]]
name = "litellm-enterprise"
version = "0.1.69"
version = "0.1.70"
source = { editable = "enterprise" }
[[package]]
name = "litellm-proxy-extras"
version = "0.4.100"
version = "0.4.101"
source = { editable = "litellm-proxy-extras" }
[[package]]

View file

@ -90,6 +90,7 @@ bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0
bedrock/eu-west-2/meta.llama3-8b-instruct-v1:0
bedrock/eu-west-2/minimax.minimax-m2.1
bedrock/eu-west-2/minimax.minimax-m2.5
bedrock/eu-west-2/nvidia.nemotron-super-3-120b
bedrock/eu-west-2/qwen.qwen3-coder-next
bedrock/eu-west-2/qwen.qwen3-next-80b-a3b
bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2
@ -234,6 +235,7 @@ bedrock/us-gov-west-1/openai.gpt-oss-120b-1:0
bedrock/us-gov-west-1/anthropic.claude-sonnet-5
bedrock/us-gov-west-1/anthropic.claude-opus-4-8
bedrock/us-gov-west-1/anthropic.claude-opus-5
bedrock/us-gov-west-1/anthropic.claude-opus-5-5
bedrock/us-gov-west-1/anthropic.claude-fable-5-1
bedrock/us-gov-east-1/nvidia.nemotron-nano-3-30b
bedrock/us-gov-east-1/nvidia.nemotron-nano-12b-v2
@ -244,5 +246,6 @@ bedrock/us-gov-east-1/openai.gpt-oss-120b-1:0
bedrock/us-gov-east-1/anthropic.claude-sonnet-5
bedrock/us-gov-east-1/anthropic.claude-opus-4-8
bedrock/us-gov-east-1/anthropic.claude-opus-5
bedrock/us-gov-east-1/anthropic.claude-opus-5-5
bedrock/us-gov-east-1/anthropic.claude-fable-5-1
bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b