Merge branch 'main' into fix/router-track-empty-model-list-live-routers

This commit is contained in:
Quinn Xu 2026-10-06 01:19:48 +08:00 • committed by GitHub
commit ba0f436635
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
1141 changed files with 106381 additions and 38507 deletions

View file

@ -38,7 +38,7 @@ commands:
parameters:
category:
type: enum
enum: ["backend", "client", "provider-harness"]
enum: ["backend", "client", "provider-harness", "redis-compat"]
default: "backend"
steps:
- run:
@ -229,6 +229,19 @@ commands:
- wait_for_service:
url: tcp://localhost:6379
timeout: "60"
install_codecov_cli:
steps:
- run:
name: Install Codecov CLI (pinned v11.3.1)
when: always
command: |
curl -sSLf -o /tmp/codecov https://cli.codecov.io/v11.3.1/linux/codecov
curl -sSLf -o /tmp/codecov.SHA256SUM https://cli.codecov.io/v11.3.1/linux/codecov.SHA256SUM
[ "$(cat /tmp/codecov.SHA256SUM)" = "ca1d64196d2d34771084afe76ea657d581bf628e31d993ff8e52ea09cc88a56d codecov" ]
(cd /tmp && sha256sum -c codecov.SHA256SUM)
chmod +x /tmp/codecov
mkdir -p "$HOME/.local/bin"
mv /tmp/codecov "$HOME/.local/bin/codecov"
start_openai_record_replay_proxy:
description: "Start the record/replay proxy (tests/_openai_record_replay_proxy.py) on host port 8090 and wait until healthy. Models whose api_base points here replay recorded provider responses, so the E2E run neither pays for nor depends on the live provider. The default upstream is OpenAI; a non-OpenAI model must point its api_base at /__recorder_upstream/<host>/ so the recorder forwards there instead of defaulting to OpenAI. Run after uv deps are synced."
steps:
@ -3308,6 +3321,169 @@ jobs:
- store_artifacts:
path: test-results
postgres_suite:
parameters:
test_path:
type: string
seed:
type: boolean
coverage_flag:
type: string
default: ""
timeout_minutes:
type: integer
machine:
image: ubuntu-2204:2024.04.1
resource_class: large
working_directory: ~/project
environment:
DATABASE_URL: postgresql://postgres:postgres@localhost:5432/litellm_test
steps:
- checkout
- skip_if_unrelated_changes
- setup_litellm_test_deps
- start_postgres:
db_name: litellm_test
image: postgres:16@sha256:e17e86066e5ef83e0952a9347f5c792b7ece00972e2aa787a6986f471b3dd3d5
- run:
name: Generate Prisma client
command: uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma
- when:
condition: << parameters.seed >>
steps:
- run:
name: Seed database schema
command: uv run --no-sync prisma db push --schema litellm/proxy/schema.prisma --accept-data-loss
- run:
name: Run << parameters.test_path >>
command: |
mkdir -p test-results
coverage_args=()
if [ -n "<< parameters.coverage_flag >>" ]; then
coverage_args=(--cov=./litellm --cov-report=xml:coverage.xml)
fi
timeout --signal=TERM << parameters.timeout_minutes >>m uv run --no-sync pytest \
<< parameters.test_path >> -vv --tb=short --durations=10 -o junit_family=xunit1 \
--junitxml=test-results/junit.xml "${coverage_args[@]}"
- when:
condition: << parameters.coverage_flag >>
steps:
- install_codecov_cli
- run:
name: Upload coverage
when: always
command: |
[ -f coverage.xml ] || { echo "no coverage.xml produced"; exit 1; }
codecov upload-process --disable-search --fail-on-error -f coverage.xml \
-F << parameters.coverage_flag >> -C "$CIRCLE_SHA1" \
-n "<< parameters.coverage_flag >>-${CIRCLE_BUILD_NUM}" --git-service github
- store_test_results:
path: test-results
- store_artifacts:
path: test-results
mcp_integration:
machine:
image: ubuntu-2204:2024.04.1
resource_class: large
working_directory: ~/project
environment:
COVERAGE_CORE: sysmon
LITELLM_LOCAL_MODEL_COST_MAP: "True"
steps:
- checkout
- skip_if_unrelated_changes
- setup_litellm_test_deps
- run:
name: Generate Prisma client
command: uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma
- run:
name: Install MCP SDK1 peer
command: |
uv venv --python 3.12 .venv-mcp-peer
uv pip install --python .venv-mcp-peer 'mcp==1.28.1' 'langchain-mcp-adapters==0.2.1'
echo "export MCP_TEST_PEER_PYTHON=$PWD/.venv-mcp-peer/bin/python" >> "$BASH_ENV"
- run:
name: Run MCP integration tests
command: |
mkdir -p test-results
env -u OPENAI_API_KEY -u ANTHROPIC_API_KEY \
timeout --signal=TERM 20m uv run --no-sync pytest \
tests/mcp_tests tests/unit/experimental_mcp_client tests/unit/proxy/_experimental/mcp_server \
tests/unit/responses/mcp --tb=short -vv --maxfail=10 -n 2 --dist=loadscope --reruns 0 \
--reruns-delay 1 --timeout=120 --rerun-except "from pytest-timeout" --durations=20 \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml:coverage.xml \
--cov-config=pyproject.toml -o junit_family=xunit1 --junitxml=test-results/junit.xml
- install_codecov_cli
- run:
name: Upload MCP integration coverage
when: always
command: |
[ -f coverage.xml ] || { echo "no coverage.xml produced"; exit 1; }
codecov upload-process --disable-search --fail-on-error -f coverage.xml \
-F mcp-integration -C "$CIRCLE_SHA1" \
-n "mcp-integration-${CIRCLE_BUILD_NUM}" --git-service github
- store_test_results:
path: test-results
- store_artifacts:
path: test-results
redis_compat:
parameters:
redis_py:
type: string
machine:
image: ubuntu-2204:2024.04.1
resource_class: large
working_directory: ~/project
steps:
- checkout
- skip_if_unrelated_changes:
category: redis-compat
- setup_litellm_test_deps
- run:
name: Pin redis-py version
command: |
uv pip install "redis==<< parameters.redis_py >>"
uv run --no-sync python -c "import redis; assert redis.__version__ == '<< parameters.redis_py >>', redis.__version__; print('redis-py', redis.__version__)"
- run:
name: Install redis-server 7.2.16
command: |
mkdir -p "$HOME/.local/bin"
cid="$(docker create redis:7.2.16@sha256:0637954999d01b7c9ce9167db2da50656e2590d3b884f1c600c5f63bb6e6773c)"
docker cp "${cid}":/usr/local/bin/redis-server "$HOME/.local/bin/redis-server"
docker rm "$cid"
redis-server --version | grep -q 'v=7.2.16'
- run:
name: Run Redis compatibility tests
command: |
mkdir -p test-results
env -u CASSETTE_REDIS_URL -u AZURE_CLIENT_ID -u AZURE_CLIENT_SECRET -u AZURE_TENANT_ID \
timeout --signal=TERM 15m uv run --no-sync pytest \
tests/unit/test_redis.py tests/unit/caching/test_redis_connection_pool.py \
tests/unit/caching/test_redis_cluster_cache.py tests/unit/caching/test_evicted_client_closer.py \
tests/local_testing/test_caching.py::test_sync_cluster_authenticates_with_azure_credentials \
tests/local_testing/test_caching.py::test_sync_cluster_authenticates_with_gcp_credentials \
--tb=short -vv --reruns 2 --reruns-delay 1 --durations=20 --cov=./litellm \
--cov-report=xml:coverage.xml -o junit_family=xunit1 --junitxml=test-results/junit.xml
- when:
condition:
equal: ["5.3.1", << parameters.redis_py >>]
steps:
- install_codecov_cli
- run:
name: Upload Redis compatibility coverage
when: always
command: |
[ -f coverage.xml ] || { echo "no coverage.xml produced"; exit 1; }
codecov upload-process --disable-search --fail-on-error -f coverage.xml \
-F redis-compat -C "$CIRCLE_SHA1" \
-n "redis-compat-${CIRCLE_BUILD_NUM}" --git-service github
- store_test_results:
path: test-results
- store_artifacts:
path: test-results
workflows:
migration_startup:
when: << pipeline.parameters.run_migration_tests >>
@ -3376,6 +3552,39 @@ workflows:
parameters:
suite: [management, database]
mode: [replica]
- postgres_suite:
name: proxy-behavior
test_path: tests/proxy_behavior
seed: true
coverage_flag: lens-postgres
timeout_minutes: 25
- postgres_suite:
name: proxy-security
test_path: tests/proxy_security_tests
seed: true
timeout_minutes: 15
- postgres_suite:
name: schema-migration
test_path: tests/proxy_migration_tests
seed: false
timeout_minutes: 20
- postgres_suite:
name: roi-database
test_path: tests/integration/database/test_roi_observed.py
seed: false
coverage_flag: roi-postgres
timeout_minutes: 10
- mcp_integration:
name: mcp-integration
- redis_compat:
name: redis-compat-<< matrix.redis_py >>
matrix:
parameters:
redis_py:
- "5.3.1"
- "6.4.0"
- "7.4.1"
- "8.0.1"
build_and_test:
unless:
or:

View file

@ -1,7 +1,7 @@
#!/usr/bin/env bash
set -uo pipefail
category="${1:?usage: classify_changes.sh <backend|client|ui|provider-harness|cost-map-only|mcp-dependencies|windows-release>}"
category="${1:?usage: classify_changes.sh <backend|client|ui|provider-harness|cost-map-only|mcp-dependencies|windows-release|redis-compat>}"
has_client=false
has_backend=false
@ -10,9 +10,14 @@ has_provider_harness=false
has_cost_map=false
has_mcp_dependencies=false
has_windows_release=false
has_redis_compat=false
outside_cost_map_set=false
while IFS= read -r file || [ -n "$file" ]; do
[ -n "$file" ] || continue
case "$file" in
litellm/_redis.py | litellm/_redis_credential_provider.py | litellm/caching/redis_cache.py | litellm/caching/evicted_client_closer.py | tests/unit/test_redis.py | tests/local_testing/test_caching.py | tests/unit/caching/test_redis_connection_pool.py | tests/unit/caching/test_redis_cluster_cache.py | tests/unit/caching/test_evicted_client_closer.py | .circleci/config.yml | .circleci/scripts/classify_changes.sh | .circleci/scripts/path_filter.sh | pyproject.toml | uv.lock)
has_redis_compat=true ;;
esac
case "$file" in
*.md | *.mdx) : ;;
pyproject.toml | */pyproject.toml | uv.lock | uv.toml | .python-version | rust-toolchain.toml | litellm-rust/* | litellm/__init__.py | litellm/proxy/proxy_server.py | litellm/*mcp* | tests/*mcp* | litellm/integrations/arize/* | tests/base_sdk_tests/* | scripts/check_mcp_sdk_install.py | .github/workflows/test-mcp-dependency-resolution.yml | .github/actions/detect-changes/* | .github/actions/setup-uv-with-retries/* | .github/actions/cache-cargo-build/* | .github/scripts/detect_changes.sh | .github/scripts/uv_sync_with_retries.sh | .circleci/scripts/classify_changes.sh | tests/unit/test_circleci_path_filter.py | tests/unit/test_detect_changes.py)
@ -54,6 +59,9 @@ case "$category" in
windows-release)
[ "$has_windows_release" = true ] && echo run || echo skip
;;
redis-compat)
[ "$has_redis_compat" = true ] && echo run || echo skip
;;
backend)
[ "$has_backend" = true ] && echo run || echo skip
;;

View file

@ -1,7 +1,7 @@
#!/usr/bin/env bash
set -uo pipefail
category="${1:?usage: path_filter.sh <backend|client|provider-harness>}"
category="${1:?usage: path_filter.sh <backend|client|provider-harness|redis-compat>}"
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
run_full() {

View file

@ -41,6 +41,7 @@ legacy_paths() {
echo tests/unit/enterprise/proxy/hooks
echo tests/unit/enterprise/proxy/management_endpoints
echo tests/unit/enterprise/proxy/test_audit_logging_endpoints.py
echo tests/unit/enterprise/proxy/test_liteadmin.py
echo tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py ;;
enterprise-routing)
echo tests/unit/google_genai

View file

@ -21,7 +21,7 @@ test_paths:
Live-provider caching cases in tests/local_testing that remain outside CI. Jobs that
glob that directory either deselect them (local_testing_part1 and part2 carry `-k "... and
not caching and not cache"`) or keep only another keyword (langfuse, router, assistants).
Separately, test-redis-compat.yml selects two IAM cluster authentication tests in
Separately, the CircleCI redis-compat jobs select two IAM cluster authentication tests in
test_caching.py by node ID. It does not run that file's other tests.
The gap was eight files and 118 tests when measured 2026-08-20; the five keyless files now
run in the caching-local shard, leaving live cases in these three. Measured 2026-08-21 with no provider

View file

@ -1,17 +0,0 @@
#!/usr/bin/env bash
set -uo pipefail
STACK_DIR="${E2E_STACK_DIR:-${RUNNER_TEMP:-/tmp}/litellm-e2e-stack}"
for pid_file in "${STACK_DIR}"/pids/*.pid; do
[[ -f "${pid_file}" ]] || continue
pkill -TERM -P "$(cat "${pid_file}")" 2>/dev/null
kill -TERM "$(cat "${pid_file}")" 2>/dev/null
rm -f "${pid_file}"
done
for container in e2e-nginx e2e-keycloak e2e-valkey e2e-jaeger e2e-postgres; do
docker rm -f "${container}" >/dev/null 2>&1
done
exit 0

View file

@ -1,83 +0,0 @@
import argparse
import os
import sys
from functools import reduce
from pathlib import Path
from typing import Final
from xml.sax.saxutils import escape
from pydantic import JsonValue, TypeAdapter, ValidationError
from secrets_to_env import MIN_MASKED_LENGTH
REDACTED: Final = "***"
json_adapter: Final[TypeAdapter[JsonValue]] = TypeAdapter(JsonValue)
def string_leaves(node: JsonValue) -> tuple[str, ...]:
match node:
case str():
return (node,)
case list():
return tuple(leaf for child in node for leaf in string_leaves(child))
case dict():
return tuple(leaf for child in node.values() for leaf in string_leaves(child))
return ()
def field_lines(value: str) -> tuple[str, ...]:
try:
return tuple(line for leaf in string_leaves(json_adapter.validate_json(value)) for line in leaf.splitlines())
except ValidationError:
return ()
def masked_values(values_files: tuple[Path, ...]) -> tuple[str, ...]:
values: Final = frozenset(
line.split("=", 1)[1].strip().strip("'")
for path in values_files
for line in path.read_text().splitlines()
if "=" in line
)
texts: Final = frozenset(text for value in values for text in (value, *field_lines(value)))
renderings: Final = frozenset(
rendering
for text in texts
if len(text) >= MIN_MASKED_LENGTH
for rendering in (text, escape(text), escape(text, {'"': "&quot;"}))
)
return tuple(sorted(renderings, key=lambda rendering: (-len(rendering), rendering)))
def redact(text: str, values: tuple[str, ...]) -> str:
return reduce(lambda redacted, value: redacted.replace(value, REDACTED), values, text)
def write_redacted(source: Path, out_dir: Path, values: tuple[str, ...]) -> None:
target: Final = out_dir / source.name
with os.fdopen(os.open(target, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600), "w") as handle:
_ = handle.write(redact(source.read_text(errors="replace"), values))
def main() -> int:
parser: Final = argparse.ArgumentParser()
_ = parser.add_argument("--values", action="append", type=Path, required=True)
_ = parser.add_argument("--out", type=Path, required=True)
_ = parser.add_argument("files", nargs="*", type=Path)
args: Final = parser.parse_args()
values_files: Final = tuple(args.values)
out_dir: Final[Path] = args.out
sources: Final = tuple(args.files)
try:
values: Final = masked_values(values_files)
out_dir.mkdir(mode=0o700, exist_ok=True)
for source in sources:
write_redacted(source, out_dir, values)
except OSError as error:
_ = sys.stderr.write(f"could not redact {error.filename}\n")
return 1
_ = sys.stdout.write(f"redacted {len(sources)} file(s) into {out_dir}\n")
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -1,55 +0,0 @@
import os
import re
import sys
from pathlib import Path
from typing import Final
from pydantic import TypeAdapter, ValidationError
secrets_adapter: Final[TypeAdapter[dict[str, str]]] = TypeAdapter(dict[str, str])
ENV_NAME: Final = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
MIN_MASKED_LENGTH: Final = 8
ACTIONS_RUNNER_FLAG: Final = "GITHUB_ACTIONS"
def main() -> int:
env_path: Final = Path(sys.argv[1])
try:
secrets: Final = {
key: value.rstrip("\r\n") for key, value in secrets_adapter.validate_json(sys.stdin.read()).items()
}
except (ValidationError, UnicodeError):
_ = sys.stderr.write("expected a JSON object containing string environment values\n")
return 1
unusable: Final = tuple(
key
for key, value in secrets.items()
if ENV_NAME.fullmatch(key) is None or any(char in value for char in "'\n\r\0")
)
if unusable:
_ = sys.stderr.write(
f"these names or values cannot be represented in both bash and dotenv: {' '.join(sorted(unusable))}\n"
)
return 1
if os.environ.get(ACTIONS_RUNNER_FLAG) == "true":
_ = sys.stdout.write(
"".join(
f"::add-mask::{value.replace('%', '%25')}\n"
for value in secrets.values()
if len(value) >= MIN_MASKED_LENGTH
)
)
sys.stdout.flush()
lines: Final = tuple(f"{key}='{value}'" for key, value in secrets.items() if value)
try:
with os.fdopen(os.open(env_path, os.O_WRONLY | os.O_APPEND | os.O_CREAT | os.O_NOFOLLOW, 0o600), "w") as handle:
os.fchmod(handle.fileno(), 0o600)
_ = handle.write("\n".join(lines) + "\n")
except OSError:
_ = sys.stderr.write("could not write the environment file\n")
return 1
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -1,52 +0,0 @@
import re
import sys
from typing import Final
SELECTABLE: Final = re.compile(r"^tests/e2e/([A-Za-z0-9_.-]+/)*test_[A-Za-z0-9_.-]+\.py$")
UNSUPPORTED: Final = re.compile(
r"^tests/e2e/(ui|claude_code|load|migrations)/"
r"|^tests/e2e/mcp/test_mcp_oauth_happy_path_e2e\.py$"
r"|^tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e\.py$"
r"|^tests/e2e/batches/test_managed_files_enforcement_e2e\.py$"
r"|^tests/e2e/guardrails/test_presidio_masking_e2e\.py$"
r"|^tests/e2e/logging/test_otel_v2_langfuse_generation_output_e2e\.py$"
r"|^tests/e2e/logging/test_langsmith_batch_serialization_e2e\.py$"
r"|^tests/e2e/logging/test_s3_log_e2e\.py$"
r"|^tests/e2e/secret_manager/"
)
HARNESS: Final = re.compile(
r"^tests/e2e/[A-Za-z0-9_.-]+\.(py|ini)$"
r"|^tests/e2e/idp_realm\.json$"
r"|^tests/e2e/management/(management_client|jwt_actors|conftest)\.py$"
r"|^tests/e2e/coverage_registry/management_cases\.py$"
r"|^tests/e2e/gateway/"
r"|^\.github/e2e-stack/"
r"|^\.github/workflows/test-e2e-changed\.yml$"
)
UNEXPANDED: Final = re.compile(r"[*?\[]")
def is_selectable(path: str) -> bool:
return SELECTABLE.match(path) is not None and UNSUPPORTED.match(path) is None
def select(changed: tuple[str, ...], canary: tuple[str, ...]) -> tuple[str, ...]:
direct: Final = frozenset(path for path in changed if is_selectable(path))
harness_changed: Final = any(HARNESS.match(path) for path in changed)
canary_tests: Final = frozenset(path for path in canary if harness_changed and is_selectable(path))
return tuple(sorted(direct | canary_tests))
def main() -> int:
canary: Final = tuple(sys.argv[1:])
unexpanded: Final = tuple(path for path in canary if UNEXPANDED.search(path))
if unexpanded:
_ = sys.stderr.write(f"the canary paths reached the selector unexpanded: {' '.join(unexpanded)}\n")
return 1
changed: Final = tuple(line.strip() for line in sys.stdin if line.strip())
_ = sys.stdout.write(" ".join(select(changed, canary)) + "\n")
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -1,231 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
umask 077
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
STACK_DIR="${E2E_STACK_DIR:-${RUNNER_TEMP:-/tmp}/litellm-e2e-stack}"
CERTS_DIR="${STACK_DIR}/certs"
LOGS_DIR="${STACK_DIR}/logs"
PIDS_DIR="${STACK_DIR}/pids"
POSTGRES_IMAGE="${E2E_POSTGRES_IMAGE:-postgres:16.6}"
VALKEY_IMAGE="${E2E_VALKEY_IMAGE:-valkey/valkey:8.1.4@sha256:81db6d39e1bba3b3ff32bd3a1b19a6d69690f94a3954ec131277b9a26b95b3aa}"
JAEGER_IMAGE="${E2E_JAEGER_IMAGE:-jaegertracing/jaeger:2.10.0}"
NGINX_IMAGE="${E2E_NGINX_IMAGE:-nginx:1.29.1-alpine@sha256:42a516af16b852e33b7682d5ef8acbd5d13fe08fecadc7ed98605ba5e3b26ab8}"
LB_PORT="${E2E_LB_PORT:-4000}"
GATEWAY_PORT_1="${E2E_GATEWAY_PORT_1:-4010}"
GATEWAY_PORT_2="${E2E_GATEWAY_PORT_2:-4011}"
BACKEND_PORT="${E2E_BACKEND_PORT:-4001}"
REDIS_PORT="${E2E_REDIS_PORT:-6379}"
DATABASE_HOST="${E2E_DATABASE_HOST:-127.0.0.1}"
DATABASE_PORT="${E2E_DATABASE_PORT:-5432}"
DATABASE_USER="${E2E_DATABASE_USER:-litellm}"
DATABASE_PASSWORD="${E2E_DATABASE_PASSWORD:-dbpassword9090}"
DATABASE_NAME="${E2E_DATABASE_NAME:-litellm}"
JAEGER_OTLP_PORT="${E2E_JAEGER_OTLP_PORT:-4318}"
JAEGER_OTLP_TLS_PORT="${E2E_JAEGER_OTLP_TLS_PORT:-4319}"
JAEGER_QUERY_PORT="${E2E_JAEGER_QUERY_PORT:-16686}"
KEYCLOAK_PORT="${E2E_KEYCLOAK_PORT:-8081}"
MASTER_KEY="${LITELLM_MASTER_KEY:-sk-e2e-$(openssl rand -hex 16)}"
mkdir -p "${CERTS_DIR}" "${LOGS_DIR}" "${PIDS_DIR}"
chmod 700 "${STACK_DIR}" "${LOGS_DIR}" "${PIDS_DIR}"
chmod 755 "${CERTS_DIR}"
log() { printf 'e2e-stack: %s\n' "$*"; }
port_open() { (exec 3<>"/dev/tcp/127.0.0.1/$1") 2>/dev/null; }
wait_for() {
local label="$1" check="$2" deadline=$((SECONDS + ${3:-120}))
until eval "${check}"; do
if ((SECONDS >= deadline)); then
log "timed out waiting for ${label}"
exit 1
fi
sleep 2
done
log "${label} is up"
}
if [[ -f "${REPO_ROOT}/tests/e2e/.env" ]]; then
set -a
source "${REPO_ROOT}/tests/e2e/.env"
set +a
fi
if [[ -z "${DD_API_KEY:-}" ]]; then
log "DD_API_KEY is empty; the gateway config enables the datadog callback, so put a Datadog API key in tests/e2e/.env"
exit 1
fi
export DD_SITE="${DD_SITE:-datadoghq.com}"
if ! port_open "${DATABASE_PORT}"; then
docker run -d --name e2e-postgres -p "${DATABASE_PORT}:5432" \
-e "POSTGRES_USER=${DATABASE_USER}" -e "POSTGRES_PASSWORD=${DATABASE_PASSWORD}" -e "POSTGRES_DB=${DATABASE_NAME}" \
"${POSTGRES_IMAGE}" >/dev/null
fi
wait_for "postgres" "port_open ${DATABASE_PORT}"
if ! port_open "${JAEGER_QUERY_PORT}"; then
docker run -d --name e2e-jaeger -p "${JAEGER_OTLP_PORT}:4318" -p "${JAEGER_QUERY_PORT}:16686" \
"${JAEGER_IMAGE}" >/dev/null
fi
wait_for "jaeger" "curl -fs http://127.0.0.1:${JAEGER_QUERY_PORT}/api/services >/dev/null"
openssl genrsa -out "${CERTS_DIR}/ca.key" 2048 2>/dev/null
openssl req -x509 -new -nodes -key "${CERTS_DIR}/ca.key" -sha256 -days 7 \
-subj "/CN=litellm-e2e-ca" \
-addext "basicConstraints=critical,CA:TRUE" -addext "keyUsage=critical,keyCertSign,cRLSign" \
-out "${CERTS_DIR}/ca.crt" 2>/dev/null
openssl genrsa -out "${CERTS_DIR}/server.key" 2048 2>/dev/null
openssl req -new -key "${CERTS_DIR}/server.key" -subj "/CN=localhost" -out "${CERTS_DIR}/server.csr" 2>/dev/null
openssl x509 -req -in "${CERTS_DIR}/server.csr" -CA "${CERTS_DIR}/ca.crt" -CAkey "${CERTS_DIR}/ca.key" \
-CAcreateserial -days 7 -sha256 \
-extfile <(printf 'basicConstraints=CA:FALSE\nkeyUsage=critical,digitalSignature,keyEncipherment\nextendedKeyUsage=serverAuth\nsubjectAltName=DNS:localhost,IP:127.0.0.1\n') \
-out "${CERTS_DIR}/server.crt" 2>/dev/null
chmod 644 "${CERTS_DIR}"/*.key "${CERTS_DIR}"/*.crt
CERTIFI_BUNDLE="$(cd "${REPO_ROOT}" && uv run --no-sync python -c 'import certifi; print(certifi.where())')"
cat "${CERTIFI_BUNDLE}" "${CERTS_DIR}/ca.crt" > "${CERTS_DIR}/ca-bundle.pem"
docker rm -f e2e-valkey >/dev/null 2>&1 || true
docker run -d --name e2e-valkey -p "${REDIS_PORT}:${REDIS_PORT}" -v "${CERTS_DIR}:/certs:ro" \
"${VALKEY_IMAGE}" valkey-server \
--cluster-enabled yes --port 0 --tls-port "${REDIS_PORT}" \
--tls-cert-file /certs/server.crt --tls-key-file /certs/server.key --tls-ca-cert-file /certs/ca.crt \
--tls-auth-clients no --cluster-announce-ip 127.0.0.1 >/dev/null
VALKEY_CLI="docker exec e2e-valkey valkey-cli --tls --cacert /certs/ca.crt -h 127.0.0.1 -p ${REDIS_PORT}"
wait_for "valkey" "${VALKEY_CLI} ping 2>/dev/null | grep -q PONG"
${VALKEY_CLI} cluster addslotsrange 0 16383 >/dev/null
wait_for "valkey cluster" "${VALKEY_CLI} cluster info 2>/dev/null | grep -q cluster_state:ok"
CONFIG_SOURCE="${REPO_ROOT}/tests/e2e/gateway/stage_mirror_ci_config.yml"
CONFIG_PATH="${CONFIG_SOURCE}"
if [[ "${REDIS_PORT}" != "6379" ]]; then
CONFIG_PATH="${STACK_DIR}/litellm-config.yml"
sed "s/port: 6379/port: ${REDIS_PORT}/" "${CONFIG_SOURCE}" > "${CONFIG_PATH}"
fi
SERVER_ENV=(
"LITELLM_MASTER_KEY=${MASTER_KEY}"
"DATABASE_HOST=${DATABASE_HOST}"
"DATABASE_PORT=${DATABASE_PORT}"
"DATABASE_USER=${DATABASE_USER}"
"DATABASE_PASSWORD=${DATABASE_PASSWORD}"
"DATABASE_NAME=${DATABASE_NAME}"
"DISABLE_SCHEMA_UPDATE=true"
"REDIS_HOST=127.0.0.1"
"REDIS_PORT=${REDIS_PORT}"
"REDIS_CLUSTER_NODES=[{\"host\":\"127.0.0.1\",\"port\":${REDIS_PORT}}]"
"CONFIG_FILE_PATH=${CONFIG_PATH}"
"STORE_MODEL_IN_DB=True"
"OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf"
"OTEL_EXPORTER_OTLP_ENDPOINT=https://127.0.0.1:${JAEGER_OTLP_TLS_PORT}"
"SSL_CERT_FILE=${CERTS_DIR}/ca-bundle.pem"
"PYTHONPATH=${REPO_ROOT}"
"JWT_PUBLIC_KEY_URL=http://127.0.0.1:${KEYCLOAK_PORT}/realms/litellm-e2e/protocol/openid-connect/certs"
"JWT_ISSUER=http://127.0.0.1:${KEYCLOAK_PORT}/realms/litellm-e2e"
"JWT_AUDIENCE=litellm-e2e"
)
if [[ -n "${VERTEXAI_CREDENTIALS:-}" ]]; then
printf '%s' "${VERTEXAI_CREDENTIALS}" > "${STACK_DIR}/vertex-adc.json"
SERVER_ENV+=("GOOGLE_APPLICATION_CREDENTIALS=${STACK_DIR}/vertex-adc.json")
fi
cd "${REPO_ROOT}"
env "${SERVER_ENV[@]}" "E2E_KEYCLOAK_PORT=${KEYCLOAK_PORT}" bash .github/e2e-stack/start-idp.sh
log "running migrations"
env "${SERVER_ENV[@]}" uv run --no-sync python migrations/run.py >"${LOGS_DIR}/migrations.log" 2>&1
start_server() {
local name="$1"; shift
env -u AWS_ROLE_NAME "${SERVER_ENV[@]}" "$@" >"${LOGS_DIR}/${name}.log" 2>&1 &
echo $! > "${PIDS_DIR}/${name}.pid"
}
if [[ "$(uname)" == "Linux" ]]; then
NGINX_UPSTREAM_HOST=127.0.0.1
NGINX_DOCKER_ARGS=(--network host)
else
NGINX_UPSTREAM_HOST=host.docker.internal
NGINX_DOCKER_ARGS=(-p "${LB_PORT}:${LB_PORT}" -p "${JAEGER_OTLP_TLS_PORT}:${JAEGER_OTLP_TLS_PORT}")
fi
cat > "${STACK_DIR}/nginx.conf" <<EOF
events {}
http {
map \$http_upgrade \$connection_upgrade {
default upgrade;
'' close;
}
upstream litellm_gateways {
server ${NGINX_UPSTREAM_HOST}:${GATEWAY_PORT_1};
server ${NGINX_UPSTREAM_HOST}:${GATEWAY_PORT_2};
}
server {
listen ${LB_PORT};
client_max_body_size 100m;
location / {
proxy_pass http://litellm_gateways;
proxy_http_version 1.1;
proxy_set_header Host \$host;
proxy_set_header X-Forwarded-For \$proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto \$scheme;
proxy_set_header Upgrade \$http_upgrade;
proxy_set_header Connection \$connection_upgrade;
proxy_buffering off;
proxy_read_timeout 600s;
proxy_send_timeout 600s;
}
}
server {
listen ${JAEGER_OTLP_TLS_PORT} ssl;
ssl_certificate /certs/server.crt;
ssl_certificate_key /certs/server.key;
client_max_body_size 100m;
location / {
proxy_pass http://${NGINX_UPSTREAM_HOST}:${JAEGER_OTLP_PORT};
}
}
}
EOF
docker rm -f e2e-nginx >/dev/null 2>&1 || true
docker run -d --name e2e-nginx "${NGINX_DOCKER_ARGS[@]}" \
-v "${STACK_DIR}/nginx.conf:/etc/nginx/nginx.conf:ro" \
-v "${CERTS_DIR}:/certs:ro" "${NGINX_IMAGE}" >/dev/null
wait_for "Jaeger OTLP TLS listener" \
"curl -sS --cacert ${CERTS_DIR}/ca.crt https://127.0.0.1:${JAEGER_OTLP_TLS_PORT}/ -o /dev/null -w '%{http_code}' | grep -qE '^[2345]'"
start_server backend uv run --no-sync uvicorn backend.main:app --host 0.0.0.0 --port "${BACKEND_PORT}"
start_server gateway-1 uv run --no-sync uvicorn gateway.main:app --workers 1 --host 0.0.0.0 --port "${GATEWAY_PORT_1}"
start_server gateway-2 uv run --no-sync uvicorn gateway.main:app --workers 1 --host 0.0.0.0 --port "${GATEWAY_PORT_2}"
wait_for "backend" "curl -fs http://127.0.0.1:${BACKEND_PORT}/health/liveliness >/dev/null" 300
wait_for "gateway-1" "curl -fs http://127.0.0.1:${GATEWAY_PORT_1}/health/liveliness >/dev/null" 300
wait_for "gateway-2" "curl -fs http://127.0.0.1:${GATEWAY_PORT_2}/health/liveliness >/dev/null" 300
wait_for "load balancer" "curl -fs http://127.0.0.1:${LB_PORT}/health/liveliness >/dev/null" 60
cat > "${STACK_DIR}/stack.env" <<EOF
LITELLM_PROXY_URL=http://127.0.0.1:${LB_PORT}
LITELLM_CONTROL_PLANE_URL=http://127.0.0.1:${BACKEND_PORT}
LITELLM_PROXY_REPLICA_URLS=http://127.0.0.1:${GATEWAY_PORT_1},http://127.0.0.1:${GATEWAY_PORT_2}
LITELLM_MASTER_KEY=${MASTER_KEY}
REDIS_HOST=127.0.0.1
REDIS_PORT=${REDIS_PORT}
E2E_OTEL_QUERY_URL=http://127.0.0.1:${JAEGER_QUERY_PORT}
E2E_OTEL_EXPORTER_ENDPOINT=https://127.0.0.1:${JAEGER_OTLP_TLS_PORT}
E2E_KEYCLOAK_URL=http://127.0.0.1:${KEYCLOAK_PORT}
E2E_KEYCLOAK_ADMIN_USER=admin
E2E_KEYCLOAK_ADMIN_PASSWORD=e2e-ephemeral-idp-not-a-secret
SSL_CERT_FILE=${CERTS_DIR}/ca-bundle.pem
DATABASE_URL=postgresql://${DATABASE_USER}:${DATABASE_PASSWORD}@${DATABASE_HOST}:${DATABASE_PORT}/${DATABASE_NAME}
EOF
log "stack is up; pytest env written to ${STACK_DIR}/stack.env"

View file

@ -22,7 +22,7 @@ TESTS_ROOT = REPO_ROOT / "tests"
ALLOWLIST_KEYS = frozenset({"description", "test_paths", "dockerfiles"})
PATH_FILTER_KEYS = frozenset({"paths", "paths-ignore"})
TEST_PATH_KEYS = frozenset({"test-path", "test-paths"})
TEST_PATH_KEYS = frozenset({"test-path", "test-paths", "test_path"})
DOCKERFILE_INPUT_KEYS = frozenset({"file", "dockerfile"})
TEST_RUNNER_RE = re.compile(r"\bpytest\b|\bcircleci tests\b|\bhelm unittest\b|\bplaywright test\b|\bpython[0-9.]*\s")
IMAGE_BUILD_RE = re.compile(r"\bdocker\s+(?:buildx\s+)?build\b")
@ -130,37 +130,24 @@ def _unit_selection_arms(repo_root: pathlib.Path = REPO_ROOT) -> Mapping[str, fr
text: Final = _uncommented(script.read_text())
return MappingProxyType(
{
label: frozenset(
match.group(0).rstrip("/") for match in TEST_TOKEN_RE.finditer(body)
)
label: frozenset(match.group(0).rstrip("/") for match in TEST_TOKEN_RE.finditer(body))
for label, body in SELECTION_ARM_RE.findall(text)
}
)
def _unit_selection_tokens(repo_root: pathlib.Path = REPO_ROOT) -> frozenset[str]:
return frozenset(
token for tokens in _unit_selection_arms(repo_root).values() for token in tokens
)
return frozenset(token for tokens in _unit_selection_arms(repo_root).values() for token in tokens)
def _wired_unit_flags(scalars: Iterable[Scalar]) -> frozenset[str]:
return frozenset(
scalar.value
for scalar in scalars
if scalar.key == "unit-flag" and "${{" not in scalar.value
)
return frozenset(scalar.value for scalar in scalars if scalar.key == "unit-flag" and "${{" not in scalar.value)
def _shard_tokens(
scalars: Iterable[Scalar], arms: Mapping[str, frozenset[str]]
) -> frozenset[str]:
def _shard_tokens(scalars: Iterable[Scalar], arms: Mapping[str, frozenset[str]]) -> frozenset[str]:
wired: Final = _wired_unit_flags(scalars)
return _invoked_test_tokens(scalars) | frozenset(
token
for label, tokens in arms.items()
if label in wired
for token in tokens
token for label, tokens in arms.items() if label in wired for token in tokens
)
@ -544,23 +531,46 @@ def _integration_groups(runner: pathlib.Path) -> dict[str, tuple[str, ...]]:
return {group: tuple(folders) for group, folders in ast.literal_eval(mapping).items()}
def _integration_github_files(runner: pathlib.Path) -> frozenset[str]:
module: Final = ast.parse(runner.read_text())
literal: Final = next(
(
node.value
for node in module.body
if isinstance(node, ast.AnnAssign)
and isinstance(node.target, ast.Name)
and node.target.id == "GITHUB_FILES"
),
None,
)
if literal is None:
return frozenset()
values: Final = literal.args[0] if isinstance(literal, ast.Call) else literal
return frozenset(ast.literal_eval(values))
def _integration_ownership(repo_root: pathlib.Path = REPO_ROOT) -> tuple[frozenset[str], tuple[Finding, ...]]:
runner: Final = repo_root / "tests/integration/run.py"
if not runner.exists():
return frozenset(), ()
groups: Final = _integration_groups(runner)
github_files: Final = _integration_github_files(runner)
integration_root: Final = repo_root / "tests/integration"
paths: Final = frozenset(
str(path.relative_to(repo_root))
for folders in groups.values()
for folder in folders
for path in (integration_root / folder).rglob("test_*.py")
if str(path.relative_to(repo_root)) not in github_files
)
browser_manifest: Final = repo_root / "tests/e2e/ui/tests/integrationCritical/expected.json"
browser_nodes: Final = json.loads(browser_manifest.read_text()) if browser_manifest.exists() else ()
browser_paths: Final = frozenset(node.split("::", 1)[0] for node in browser_nodes)
circle_path: Final = repo_root / ".circleci/config.yml"
circle: Final = yaml.safe_load(circle_path.read_text()) if circle_path.exists() else {}
circle_test_path_tokens: Final = _invoked_test_tokens(
scalar for scalar in _scalars(circle, "config.yml") if scalar.key == "test_path"
)
steps: Final = circle.get("jobs", {}).get("integration_contracts", {}).get("steps", ())
invoked: Final = any(
".circleci/scripts/run_integration.sh" in scalar.value
@ -595,10 +605,22 @@ def _integration_ownership(repo_root: pathlib.Path = REPO_ROOT) -> tuple[frozens
for path in (repo_root / ".github/workflows").glob("*.y*ml")
for scalar in _scalars(yaml.safe_load(path.read_text()), path.name)
)
findings: Final = tuple(
Finding(path, "integration contract is also selected by GitHub Actions")
for path in paths
if any(_token_covers(token, path) for token in gha_tokens)
findings: Final = (
tuple(
Finding(path, "integration contract is also selected by GitHub Actions")
for path in paths
if any(_token_covers(token, path) for token in gha_tokens)
)
+ tuple(
Finding(path, "GitHub-owned integration contract has no invoking job")
for path in sorted(github_files)
if not any(_token_covers(token, path) for token in gha_tokens | circle_test_path_tokens)
)
+ tuple(
Finding(path, "GitHub-owned integration file is missing")
for path in sorted(github_files)
if not (repo_root / path).is_file()
)
)
browser_commands: Final = tuple(
scalar.value
@ -642,7 +664,7 @@ def _integration_ownership(repo_root: pathlib.Path = REPO_ROOT) -> tuple[frozens
return frozenset(), findings + (
Finding(str(runner.relative_to(repo_root)), "dedicated CircleCI runner is missing"),
)
return paths | browser_paths, findings + group_findings + browser_findings + exclusion_findings
return paths | browser_paths | github_files, findings + group_findings + browser_findings + exclusion_findings
def main() -> int:

View file

@ -79,12 +79,6 @@ on:
description: "Unique name for the coverage artifact (must be unique per run)"
required: true
type: string
legacy-mcp-peer:
description: "Install the isolated SDK1 peer for MCP compatibility tests"
required: false
type: boolean
default: false
permissions:
contents: read
@ -142,17 +136,10 @@ jobs:
- name: Install dependencies
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: 8
env:
LEGACY_MCP_PEER: ${{ inputs.legacy-mcp-peer }}
run: |
diff -u model_prices_and_context_window.json litellm/model_prices_and_context_window_backup.json
.github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router --extra saml
uv run --no-sync python -c 'import os, sys; print(sys.version); assert f"{sys.version_info.major}.{sys.version_info.minor}" == os.environ["UV_PYTHON"]'
if [ "$LEGACY_MCP_PEER" = "true" ]; then
uv venv --python "${UV_PYTHON}" .venv-mcp-peer
uv pip install --python .venv-mcp-peer 'mcp==1.28.1' 'langchain-mcp-adapters==0.2.1'
echo "MCP_TEST_PEER_PYTHON=$GITHUB_WORKSPACE/.venv-mcp-peer/bin/python" >> "$GITHUB_ENV"
fi
- name: Cache Prisma binaries
if: steps.changes.outputs.decision != 'skip'

View file

@ -15,6 +15,9 @@ on:
- gateway/main.py
- backend/Dockerfile
- backend/main.py
- deploy/lens/**
- litellm/proxy/lens/**
- tests/e2e/migrations/lens_compose_smoke.sh
- docker/component_entrypoint.sh
- docker/entrypoint.sh
- litellm/proxy/prisma_migration.py
@ -37,6 +40,80 @@ concurrency:
cancel-in-progress: true
jobs:
lens-worker-image:
name: lens-worker-image (${{ matrix.arch }})
runs-on: ${{ matrix.runner }}
if: >-
github.event_name != 'pull_request' ||
github.event.pull_request.head.repo.full_name == github.repository
timeout-minutes: 15
permissions:
contents: read
strategy:
fail-fast: false
matrix:
include:
- arch: amd64
runner: ubuntu-latest
grype_sha256: edda0968d8827daab01d32b3cd7de192ae0915005e7bbfcfef9e68e79bc43343
- arch: arm64
runner: ubuntu-24.04-arm
grype_sha256: 553e4c36d9d61349830ba6034d43b8700a7f10576d3e2f4981c0fd2b96086465
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Build the release worker
env:
RELEASE_TAG: sha-${{ github.sha }}
run: docker build --build-arg LITELLM_RELEASE_TAG="${RELEASE_TAG}" -f deploy/lens/Dockerfile -t lens-worker-scan .
- name: Verify the standalone worker on a read-only filesystem
env:
RELEASE_TAG: sha-${{ github.sha }}
run: |
docker run --rm --network none --read-only --cap-drop ALL \
--tmpfs /tmp:rw,noexec,nosuid,size=1g --security-opt no-new-privileges \
-e EXPECTED_RELEASE_TAG="${RELEASE_TAG}" --entrypoint python lens-worker-scan -c '
import os
import lens.worker
from lens.release import release_tag
from lens.trace_store import trace_store
assert os.getuid() == 65532
assert release_tag() == os.environ["EXPECTED_RELEASE_TAG"]
with trace_store() as store:
assert store.count() == 0
'
- name: Reject a dependency whose hash has changed
run: |
docker build --target builder -f deploy/lens/Dockerfile -t lens-worker-deps .
sed -E 's/sha256:[0-9a-f]{64}/sha256:0000000000000000000000000000000000000000000000000000000000000000/g' \
deploy/lens/requirements.lock > "$RUNNER_TEMP/tampered.lock"
if docker run --rm -v "$RUNNER_TEMP/tampered.lock:/tmp/tampered.lock:ro" \
--entrypoint uv lens-worker-deps pip sync --python /app/.venv/bin/python \
--require-hashes --only-binary :all: --reinstall --no-cache /tmp/tampered.lock \
> "$RUNNER_TEMP/hash-check.log" 2>&1; then
echo "::error::Dependency hash mismatch was accepted"
exit 1
fi
cat "$RUNNER_TEMP/hash-check.log"
grep -qi 'hash mismatch' "$RUNNER_TEMP/hash-check.log"
- name: Download Grype v0.114.0
env:
ARCH: ${{ matrix.arch }}
GRYPE_SHA256: ${{ matrix.grype_sha256 }}
run: |
curl -fsSL --retry 3 -o "$RUNNER_TEMP/grype.tar.gz" \
"https://github.com/anchore/grype/releases/download/v0.114.0/grype_0.114.0_linux_${ARCH}.tar.gz"
echo "${GRYPE_SHA256} $RUNNER_TEMP/grype.tar.gz" | sha256sum -c -
tar xzf "$RUNNER_TEMP/grype.tar.gz" -C "$RUNNER_TEMP" grype
chmod +x "$RUNNER_TEMP/grype"
- name: Scan the worker for fixable HIGH/CRITICAL CVEs
env:
GRYPE_MATCH_PYTHON_USING_CPES: "true"
run: |
"$RUNNER_TEMP/grype" lens-worker-scan \
--config .grype.yaml --only-fixed --fail-on high --output table
image-scan:
name: image-scan
runs-on: ubuntu-latest
@ -113,7 +190,7 @@ jobs:
persist-credentials: false
- name: Build runtime image
run: docker build -f Dockerfile -t litellm-runtime-scan:${{ github.sha }} .
run: docker build --build-arg LITELLM_RELEASE_TAG=v0.0.0-lens-ci -f Dockerfile -t litellm-runtime-scan:${{ github.sha }} .
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
@ -127,6 +204,11 @@ jobs:
python -m pip install "pytest==9.0.3"
python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v
- name: Verify the bundled Lens Compose installation and restart
env:
LITELLM_IMAGE: litellm-runtime-scan:${{ github.sha }}
run: bash tests/e2e/migrations/lens_compose_smoke.sh
migrations-image:
name: migrations-image
runs-on: ubuntu-latest

View file

@ -34,7 +34,14 @@ jobs:
with:
persist-credentials: false
- name: Build Lens worker
run: docker build -f deploy/lens/Dockerfile -t lens-worker:${{ github.sha }} .
run: docker build --build-arg LITELLM_RELEASE_TAG=sha-${{ github.sha }} -f deploy/lens/Dockerfile -t lens-worker:${{ github.sha }} .
- name: Reject custom builds without a matching release tag
run: |
if docker build --progress plain -f deploy/lens/Dockerfile -t lens-worker:unversioned . > missing-tag.log 2>&1; then
echo "::error::An unversioned worker build unexpectedly succeeded"
exit 1
fi
grep -F 'LITELLM_RELEASE_TAG: Pass --build-arg LITELLM_RELEASE_TAG matching the gateway' missing-tag.log
- name: Verify standalone imports with a read-only filesystem
run: |
docker run --rm --network none --read-only --cap-drop ALL --tmpfs /tmp:rw,noexec,nosuid,size=1g \
@ -54,11 +61,11 @@ jobs:
-v "$PWD/tests/proxy_behavior/lens/worker_storage_smoke.py:/app/storage_smoke.py:ro" \
--entrypoint python lens-worker:${{ github.sha }} /app/storage_smoke.py
- name: Publish versioned Lens worker
if: github.event_name != 'pull_request' && github.repository == 'BerriAI/litellm'
if: github.event_name != 'pull_request' && github.repository == 'BerriAI/litellm' && github.ref == 'refs/heads/main'
env:
REGISTRY_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REGISTRY_USER: ${{ github.actor }}
IMAGE: ghcr.io/berriai/litellm-lens-worker:sha-${{ github.sha }}
IMAGE: ghcr.io/berriai/litellm-lens-worker-dev:sha-${{ github.sha }}
run: |
printf '%s' "$REGISTRY_TOKEN" | docker login ghcr.io -u "$REGISTRY_USER" --password-stdin
docker tag lens-worker:${{ github.sha }} "$IMAGE"

View file

@ -1,265 +0,0 @@
name: e2e-changed-tests
on:
pull_request:
concurrency:
group: e2e-changed-${{ github.event.pull_request.number }}
cancel-in-progress: true
permissions: {}
jobs:
detect:
name: Detect changed e2e tests
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
contents: read
pull-requests: read
outputs:
tests: ${{ steps.changed.outputs.tests }}
any: ${{ steps.changed.outputs.any }}
steps:
- name: Checkout the selector and the canary suite
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
sparse-checkout: |
.github/e2e-stack
tests/e2e/access_control
tests/e2e/management/test_jwt_management_e2e.py
tests/e2e/other/test_jwt_auth_e2e.py
persist-credentials: false
ref: ${{ github.sha }}
- name: List the e2e test files this PR added or modified
id: changed
env:
GH_TOKEN: ${{ github.token }}
REPO: ${{ github.repository }}
PR_NUMBER: ${{ github.event.pull_request.number }}
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: |
gh api "repos/${REPO}/pulls/${PR_NUMBER}" \
--jq 'select(.head.sha == env.HEAD_SHA and .changed_files < 3000) | .head.sha' \
| grep -Fxq "${HEAD_SHA}"
files="$(gh api "repos/${REPO}/pulls/${PR_NUMBER}/files" --paginate \
--jq '.[] | select(.status != "removed") | .filename')"
gh api "repos/${REPO}/pulls/${PR_NUMBER}" --jq '.head.sha' | grep -Fxq "${HEAD_SHA}"
tests="$(printf '%s\n' "${files}" \
| python3 .github/e2e-stack/select_tests.py tests/e2e/access_control/test_*.py \
tests/e2e/management/test_jwt_management_e2e.py tests/e2e/other/test_jwt_auth_e2e.py)"
echo "tests=${tests}" >> "${GITHUB_OUTPUT}"
if [ -n "${tests}" ]; then
echo "any=true" >> "${GITHUB_OUTPUT}"
echo "selected e2e tests: ${tests}"
else
echo "any=false" >> "${GITHUB_OUTPUT}"
echo "no changed e2e test files supported by this stack; nothing to run"
fi
run:
name: Run changed e2e tests against the stage-mirror stack
needs: detect
if: needs.detect.outputs.any == 'true' && github.event.pull_request.head.repo.full_name == github.repository
runs-on: ubuntu-latest
timeout-minutes: 90
environment: e2e-changed
permissions:
contents: read
id-token: write
services:
postgres:
image: postgres:16.6
env:
POSTGRES_USER: litellm
POSTGRES_PASSWORD: dbpassword9090
POSTGRES_DB: litellm
ports:
- 5432:5432
options: >-
--health-cmd "pg_isready -U litellm"
--health-interval 5s
--health-timeout 5s
--health-retries 10
jaeger:
image: jaegertracing/jaeger:2.10.0
ports:
- 4318:4318
- 16686:16686
steps:
- name: Validate configuration
env:
ROLE: ${{ vars.E2E_AWS_ROLE_TO_ASSUME }}
run: test -n "${ROLE}" || { echo "::error::Set repo variable E2E_AWS_ROLE_TO_ASSUME to an OIDC role with read access to the e2e secrets"; exit 1; }
- name: Checkout
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
ref: ${{ github.sha }}
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.13"
- name: Set up uv
uses: ./.github/actions/setup-uv-with-retries
with:
version: "0.10.9"
- name: Cache the Rust build
uses: ./.github/actions/cache-cargo-build
- name: Install dependencies
run: |
.github/scripts/uv_sync_with_retries.sh --frozen \
--extra proxy --extra proxy-runtime --extra extra_proxy \
--extra semantic-router --extra bedrock-realtime \
--group ci --group proxy-dev --group e2e-dev
uv pip install "pipecat-ai[openai]==1.4.0"
- name: Cache Prisma binaries
uses: ./.github/actions/cache-prisma-binaries
- name: Generate Prisma client
run: uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma
- name: Install Playwright chromium
run: uv run --no-sync playwright install --with-deps chromium
- name: Configure AWS credentials
id: aws
uses: aws-actions/configure-aws-credentials@e7f100cf4c008499ea8adda475de1042d6975c7b # v6.2.0
with:
role-to-assume: ${{ vars.E2E_AWS_ROLE_TO_ASSUME }}
aws-region: us-east-1
role-session-name: litellm-e2e-changed-${{ github.run_id }}
role-duration-seconds: 900
output-env-credentials: false
output-credentials: true
- name: Fetch provider credentials from AWS Secrets Manager
env:
AWS_ACCESS_KEY_ID: ${{ steps.aws.outputs.aws-access-key-id }}
AWS_SECRET_ACCESS_KEY: ${{ steps.aws.outputs.aws-secret-access-key }}
AWS_SESSION_TOKEN: ${{ steps.aws.outputs.aws-session-token }}
AWS_DEFAULT_REGION: us-east-1
run: |
umask 077
aws secretsmanager get-secret-value --secret-id litellm-e2e-changed-provider-keys \
--query SecretString --output text \
| uv run --no-sync python .github/e2e-stack/secrets_to_env.py tests/e2e/.env
aws secretsmanager get-secret-value --secret-id litellm-e2e-changed-license \
--query SecretString --output text \
| jq -R -s '{"LITELLM_LICENSE": .}' \
| uv run --no-sync python .github/e2e-stack/secrets_to_env.py tests/e2e/.env
- name: Boot the stage-mirror stack
id: boot
run: |
umask 077
if ! bash .github/e2e-stack/up.sh > "${RUNNER_TEMP}/e2e-boot.log" 2>&1; then
echo "::error::stage-mirror stack failed to boot; raw logs are not published"
exit 1
fi
- name: Export stack environment
run: |
master_key="$(grep '^LITELLM_MASTER_KEY=' "${RUNNER_TEMP}/litellm-e2e-stack/stack.env" | cut -d= -f2-)"
echo "::add-mask::${master_key}"
cat "${RUNNER_TEMP}/litellm-e2e-stack/stack.env" >> "${GITHUB_ENV}"
- name: Run the selected tests three times
env:
TESTS: ${{ needs.detect.outputs.tests }}
E2E_FIXTURE_MODE: live
E2E_PROVIDER_EDGE_HOST_REACHABLE: '1'
E2E_OWNED_GATEWAY: '1'
COLUMNS: '400'
run: |
umask 077
read -r -a test_files <<< "${TESTS}"
for pass in 1 2 3; do
report="${RUNNER_TEMP}/e2e-pass-${pass}.xml"
log="${RUNNER_TEMP}/e2e-pass-${pass}.log"
echo "::group::pass ${pass} of 3"
set +e
uv run --no-sync pytest "${test_files[@]}" --rootdir=. -v --reruns 0 -p no:cacheprovider \
-o junit_family=xunit1 --junitxml="${report}" > "${log}" 2>&1
status=$?
uv run --no-sync python .github/e2e-stack/assert_tests_ran.py "${report}" "${test_files[@]}"
verified=$?
set -e
grep -E '^(FAILED|ERROR) ' "${log}" || true
grep -E '^=+ .* in [0-9.]+s( \([0-9:]+\))? =+$' "${log}" | tail -n 1
echo "::endgroup::"
if [ "${status}" = "5" ]; then
echo "::error::the selected files collected no runnable tests, so nothing was verified"
exit 1
fi
if [ "${status}" != "0" ]; then
echo "::error::pass ${pass} of 3 failed with exit code ${status}"
exit "${status}"
fi
if [ "${verified}" != "0" ]; then
echo "::error::pass ${pass} of 3 did not verify every selected file"
exit 1
fi
echo "pass ${pass} of 3 passed"
done
- name: Redact the pytest output
if: always() && steps.boot.outcome == 'success'
run: |
umask 077
shopt -s nullglob
uv run --no-sync python .github/e2e-stack/redact_output.py \
--values tests/e2e/.env --values "${RUNNER_TEMP}/litellm-e2e-stack/stack.env" \
--out "${RUNNER_TEMP}/e2e-redacted" "${RUNNER_TEMP}"/e2e-pass-*.log "${RUNNER_TEMP}"/e2e-pass-*.xml
- name: Keep the redacted pytest output
if: always() && steps.boot.outcome == 'success'
uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1
with:
name: e2e-changed-pytest-output-${{ github.run_attempt }}
path: ${{ runner.temp }}/e2e-redacted
retention-days: 14
if-no-files-found: ignore
- name: Stop the stack
if: always() && steps.boot.outcome != 'skipped'
run: bash .github/e2e-stack/down.sh
- name: Remove credentials and raw output
if: always()
run: |
rm -f tests/e2e/.env "${RUNNER_TEMP}/e2e-boot.log" "${RUNNER_TEMP}"/e2e-pass-*.log "${RUNNER_TEMP}"/e2e-pass-*.xml
rm -rf "${RUNNER_TEMP}/litellm-e2e-stack" "${RUNNER_TEMP}/e2e-redacted"
gate:
name: e2e-changed-tests
needs: [detect, run]
if: always()
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- name: Require three successful passes when tests changed
env:
DETECT_RESULT: ${{ needs.detect.result }}
ANY_TESTS: ${{ needs.detect.outputs.any }}
RUN_RESULT: ${{ needs.run.result }}
run: |
if [ "${DETECT_RESULT}" != "success" ]; then
echo "::error::changed-test detection did not succeed"
exit 1
fi
if [ "${ANY_TESTS}" = "false" ]; then
echo "no changed e2e test files supported by this stack; nothing to run"
exit 0
fi
if [ "${ANY_TESTS}" != "true" ] || [ "${RUN_RESULT}" != "success" ]; then
echo "::error::selected e2e tests require an approved, successful run; fork PRs must run from a reviewed same-repository branch"
exit 1
fi

View file

@ -5,6 +5,27 @@ on:
branches:
- main
- "litellm_**"
paths:
- "**/pyproject.toml"
- "uv.lock"
- "uv.toml"
- ".python-version"
- "rust-toolchain.toml"
- "litellm-rust/**"
- "litellm/__init__.py"
- "litellm/proxy/proxy_server.py"
- "litellm/**/*mcp*"
- "litellm/**/*mcp*/**"
- "litellm/integrations/arize/**"
- "scripts/check_mcp_sdk_install.py"
- "tests/base_sdk_tests/**"
- ".github/workflows/test-mcp-dependency-resolution.yml"
- ".github/actions/detect-changes/**"
- ".github/actions/setup-uv-with-retries/**"
- ".github/actions/cache-cargo-build/**"
- ".github/scripts/detect_changes.sh"
- ".github/scripts/uv_sync_with_retries.sh"
- ".circleci/scripts/classify_changes.sh"
permissions:
contents: read

View file

@ -1,167 +0,0 @@
name: "Postgres Tests"
on:
pull_request:
branches:
- main
- "litellm_**"
push:
branches:
- main
workflow_dispatch:
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.sha }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
postgres:
name: ${{ matrix.shard }}
runs-on: ubuntu-latest
timeout-minutes: ${{ matrix.job-timeout-minutes }}
permissions:
contents: read
id-token: write
services:
postgres:
image: postgres:16@sha256:e17e86066e5ef83e0952a9347f5c792b7ece00972e2aa787a6986f471b3dd3d5
env:
POSTGRES_USER: postgres
POSTGRES_PASSWORD: postgres
POSTGRES_DB: litellm_test
ports:
- 5432:5432
options: >-
--health-cmd pg_isready
--health-interval 10s
--health-timeout 5s
--health-retries 10
strategy:
fail-fast: false
matrix:
include:
- shard: proxy-behavior
test-path: "tests/proxy_behavior"
seed: db-push
workers: 0
timeout-minutes: 25
job-timeout-minutes: 50
- shard: proxy-security
test-path: "tests/proxy_security_tests"
seed: db-push
workers: 0
timeout-minutes: 15
job-timeout-minutes: 40
- shard: schema-migration
test-path: "tests/proxy_migration_tests"
seed: none
workers: 0
timeout-minutes: 20
job-timeout-minutes: 45
env:
DATABASE_URL: "postgresql://postgres:postgres@localhost:5432/litellm_test"
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
timeout-minutes: 3
with:
persist-credentials: false
- name: Detect relevant changes
id: changes
timeout-minutes: 2
uses: ./.github/actions/detect-changes
- name: Set up Python
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: 3
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Set up uv
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: 3
uses: ./.github/actions/setup-uv-with-retries
with:
version: "0.10.9"
- name: Cache uv dependencies
if: steps.changes.outputs.decision != 'skip' && github.ref == 'refs/heads/main'
timeout-minutes: 5
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/uv
.venv
key: ${{ runner.os }}-uv-postgres-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-postgres-
- name: Cache uv dependencies
if: steps.changes.outputs.decision != 'skip' && github.ref != 'refs/heads/main'
timeout-minutes: 5
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/uv
.venv
key: ${{ runner.os }}-uv-postgres-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-postgres-
- name: Install dependencies
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: 12
run: |
.github/scripts/uv_sync_with_retries.sh --frozen --all-groups --all-extras
- name: Cache Prisma binaries
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: 3
uses: ./.github/actions/cache-prisma-binaries
- name: Generate Prisma client
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: 5
run: |
uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma
- name: Seed database schema
if: steps.changes.outputs.decision != 'skip' && matrix.seed != 'none'
timeout-minutes: 10
run: |
uv run --no-sync prisma db push --schema litellm/proxy/schema.prisma --accept-data-loss
- name: Run tests
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: ${{ matrix.timeout-minutes }}
env:
TEST_PATH: ${{ matrix.test-path }}
WORKERS: ${{ matrix.workers }}
PYTEST_ADDOPTS: ${{ matrix.shard == 'proxy-behavior' && '--cov=./litellm --cov-report=xml:coverage-lens-postgres.xml' || '' }}
run: |
if [ "${WORKERS}" = "0" ]; then
uv run --no-sync pytest ${TEST_PATH:?} -vv --tb=short --durations=10
else
uv run --no-sync pytest ${TEST_PATH:?} -vv --tb=short --durations=10 -n "${WORKERS}"
fi
- name: Upload Lens database coverage
if: steps.changes.outputs.decision != 'skip' && matrix.shard == 'proxy-behavior' && !cancelled()
uses: codecov/codecov-action@303a32d7a59b442fa8d48b6a1cc6825c09c847a5 # v7.1.1
with:
use_oidc: true
version: v11.3.1
root_dir: ${{ github.workspace }}
files: coverage-lens-postgres.xml
flags: lens-postgres
fail_ci_if_error: true

View file

@ -1,106 +0,0 @@
name: "Unit Tests: Redis Client Version Compatibility"
on:
pull_request:
branches:
- main
- "litellm_**"
paths:
- "litellm/_redis.py"
- "litellm/_redis_credential_provider.py"
- "litellm/caching/redis_cache.py"
- "litellm/caching/evicted_client_closer.py"
- "tests/unit/test_redis.py"
- "tests/local_testing/test_caching.py"
- "tests/unit/caching/test_redis_connection_pool.py"
- "tests/unit/caching/test_redis_cluster_cache.py"
- "tests/unit/caching/test_evicted_client_closer.py"
- ".github/workflows/test-redis-compat.yml"
- "pyproject.toml"
- "uv.lock"
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
redis-compat:
name: "redis-py ${{ matrix.redis-version }}"
runs-on: ubuntu-latest
timeout-minutes: 15
permissions:
contents: read
id-token: write
strategy:
fail-fast: false
matrix:
# 5.3.1 is the version pinned in uv.lock (redisvl caps it below 6); the
# newer legs prove the inspect.signature introspection in litellm/_redis.py
# keeps extracting kwargs on the redis-py releases people actually run now.
# Only the exact release 6.0.0 is skipped: rq (pulled by the proxy extra)
# specifies `redis != 6`, which excludes 6.0.0 alone, so 6.4.0 stands in
# for the 6.x line.
redis-version: ["5.3.1", "6.4.0", "7.4.1", "8.0.1"]
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Set up uv
uses: ./.github/actions/setup-uv-with-retries
with:
version: "0.10.9"
- name: Install dependencies
run: |
.github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra extra_proxy --extra semantic-router
- name: Pin redis-py to the matrix version
env:
REDIS_VERSION: ${{ matrix.redis-version }}
run: |
uv pip install "redis==${REDIS_VERSION:?}"
uv run --no-sync python -c "import redis; assert redis.__version__ == '${REDIS_VERSION:?}', redis.__version__; print('redis-py', redis.__version__)"
- name: Build Redis for cluster authentication tests
run: |
curl --fail --location --retry 3 https://download.redis.io/releases/redis-7.2.16.tar.gz -o "$RUNNER_TEMP/redis-7.2.16.tar.gz"
echo "960a8ec15e34ff40e57ff16837b26b33bd81f2da6d24497bb63de532a323a18e $RUNNER_TEMP/redis-7.2.16.tar.gz" | sha256sum --check
tar -xzf "$RUNNER_TEMP/redis-7.2.16.tar.gz" -C "$RUNNER_TEMP"
make -C "$RUNNER_TEMP/redis-7.2.16" -j2 MALLOC=libc OPTIMIZATION=-O1 redis-server
echo "$RUNNER_TEMP/redis-7.2.16/src" >> "$GITHUB_PATH"
- name: Run redis unit tests
run: |
redis-server --version
uv run --no-sync pytest \
tests/unit/test_redis.py \
tests/unit/caching/test_redis_connection_pool.py \
tests/unit/caching/test_redis_cluster_cache.py \
tests/unit/caching/test_evicted_client_closer.py \
tests/local_testing/test_caching.py::test_sync_cluster_authenticates_with_azure_credentials \
tests/local_testing/test_caching.py::test_sync_cluster_authenticates_with_gcp_credentials \
--tb=short -vv \
--reruns 2 \
--reruns-delay 1 \
--durations=20 \
--cov=./litellm --cov-report=xml:coverage-redis.xml
- name: Upload Redis coverage
if: matrix.redis-version == '5.3.1'
uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5
with:
use_oidc: true
files: coverage-redis.xml
flags: redis-compat
fail_ci_if_error: false

View file

@ -50,18 +50,9 @@ jobs:
fail-fast: false
matrix:
include:
- shard: mcp-integration
artifact-name: mcp-integration
test-path: "tests/mcp_tests"
unit-flag: mcp-integration
workers: 2
reruns: 0
timeout-minutes: 20
job-timeout-minutes: 60
- shard: core-utils
artifact-name: core-utils
test-path: ""
test-path: tests/unit/decisions
unit-flag: core-utils
workers: 2
reruns: 1
@ -141,6 +132,7 @@ jobs:
artifact-name: proxy-endpoints
test-path: >-
tests/unit/proxy/analytics_endpoints
tests/unit/proxy/decisions_endpoints
tests/unit/proxy/management_endpoints
tests/unit/proxy/list_api
tests/unit/proxy/memory
@ -301,4 +293,3 @@ jobs:
timeout-minutes: ${{ matrix.timeout-minutes }}
job-timeout-minutes: ${{ matrix.job-timeout-minutes }}
artifact-name: ${{ matrix.artifact-name }}
legacy-mcp-peer: ${{ matrix.shard == 'mcp-integration' }}

3
.gitignore vendored
View file

@ -151,3 +151,6 @@ litellm.log
.coverage-rust
coverage-rust.xml
# make lens-dev worker token, generated config and logs
.lens-dev/

View file

@ -114,8 +114,20 @@ RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_BUILD_IMAGE AS liteadmin-builder
COPY --from=uvbin /uv /usr/local/bin/uv
RUN apk add --no-cache python-3.13
ADD --checksum=sha256:2f7ae5cdd9d91731c0990e74a58239dc3e3fd2bf28dab23b55eafcdc47aaf87e \
https://github.com/BerriAI/litellm-admin-agent/archive/ef501e94bc9fbacb9233b922abf71427f030408c.tar.gz /tmp/liteadmin.tar.gz
RUN mkdir /tmp/liteadmin && tar xzf /tmp/liteadmin.tar.gz --strip-components=1 -C /tmp/liteadmin && \
uv venv /opt/liteadmin --python python3.13 && \
uv pip install --python /opt/liteadmin/bin/python --require-hashes -r /tmp/liteadmin/requirements.txt && \
uv pip install --python /opt/liteadmin/bin/python --no-deps /tmp/liteadmin
# Runtime stage
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG}
USER root
@ -141,6 +153,7 @@ ENV PATH="/app/.venv/bin:${PATH}" \
# ship (manifest-scanning tools attribute everything in it to this image).
# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path.
COPY --from=builder /app/.venv /app/.venv
COPY --from=liteadmin-builder /opt/liteadmin /opt/liteadmin
COPY --from=builder /app/docker /app/docker
COPY --from=builder /app/schema.prisma /app/schema.prisma
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py

View file

@ -4,7 +4,7 @@
.PHONY: help test test-unit test-unit-llms test-unit-proxy-guardrails test-unit-proxy-core test-unit-proxy-misc test-unit-proxy-root \
test-unit-integrations test-unit-core-utils test-unit-other test-unit-root \
test-proxy-unit-a test-proxy-unit-b test-integration test-unit-helm \
test-rust-extension rust-sqlx-prepare \
test-rust-extension rust-sqlx-prepare lens-dev \
info lint lint-inner lint-dev lint-checks format \
lint-basedpyright lint-e2e-basedpyright lint-basedpyright-budget-update lint-type-discipline lint-type-discipline-budget-update \
lint-ruff-budget lint-ruff-budget-update lint-budget-update lint-gate \
@ -58,6 +58,7 @@ help:
@echo " make test-unit-helm - Run helm unit tests"
@echo " make test-rust-extension - Build the Rust extension and run its public Python tests"
@echo " make rust-sqlx-prepare - Refresh litellm-rust/crates/db/.sqlx against a migrated Postgres container"
@echo " make lens-dev - Run proxy + Lens worker + hot-reload dashboard (ARGS=\"--seed large --seed-logs\", LENS_DEV_PROXY_PORT, LENS_DEV_UI_PORT)"
@echo ""
@echo "Heavy targets (check, lint) queue for LITELLM_GATE_SLOTS machine-wide"
@echo "slots (default 2; 0 disables) so parallel sessions don't thrash one machine."
@ -311,6 +312,9 @@ test-rust-extension:
rust-sqlx-prepare:
cd litellm-rust && cargo run -p litellm-db-testing --bin sqlx-prepare
lens-dev:
./scripts/lens_dev.sh $(ARGS)
test: install-test-deps
$(UV_RUN) pytest tests/

View file

@ -390,11 +390,13 @@ Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call th
| [Sail (`sail`)](https://docs.litellm.ai/docs/providers/sail) | ✅ | ✅ | ✅ | | | | | | | |
| [Sambanova (`sambanova`)](https://docs.litellm.ai/docs/providers/sambanova) | ✅ | ✅ | ✅ | | | | | | | |
| [Snowflake (`snowflake`)](https://docs.litellm.ai/docs/providers/snowflake) | ✅ | ✅ | ✅ | | | | | | | |
| [Strands Decider (`strands_decider`)](https://docs.litellm.ai/docs/providers) | | | | | | | | | | |
| [Text Completion Codestral (`text-completion-codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |
| [Text Completion OpenAI (`text-completion-openai`)](https://docs.litellm.ai/docs/providers/text_completion_openai) | ✅ | ✅ | ✅ | | | ✅ | ✅ | ✅ | ✅ | |
| [Together AI (`together_ai`)](https://docs.litellm.ai/docs/providers/togetherai) | ✅ | ✅ | ✅ | | | | | | | |
| [Topaz (`topaz`)](https://docs.litellm.ai/docs/providers/topaz) | ✅ | ✅ | ✅ | | | | | | | |
| [Triton (`triton`)](https://docs.litellm.ai/docs/providers/triton-inference-server) | ✅ | ✅ | ✅ | | | | | | | |
| [Typesafe Decisions API (`typesafe`)](https://docs.litellm.ai/docs/providers) | | | | | | | | | | |
| [V0 (`v0`)](https://docs.litellm.ai/docs/providers/v0) | ✅ | ✅ | ✅ | | | | | | | |
| [Vercel AI Gateway (`vercel_ai_gateway`)](https://docs.litellm.ai/docs/providers/vercel_ai_gateway) | ✅ | ✅ | ✅ | | | | | | | |
| [VLLM (`vllm`)](https://docs.litellm.ai/docs/providers/vllm) | ✅ | ✅ | ✅ | | | | | | | |

View file

@ -71,6 +71,8 @@ RUN sed -i 's/\r$//' docker/component_entrypoint.sh && chmod +x docker/component
# ---------- Runtime ----------
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG}
USER root

View file

@ -22,6 +22,7 @@ BACKEND_PATH_PREFIXES: tuple[str, ...] = (
"/customer/",
"/end_user/",
"/sso/",
"/liteadmin/slack/connect/",
"/login",
"/v2/login",
"/v3/login",

View file

@ -5710,6 +5710,17 @@
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "histogram_quantile(0.95, sum(rate(litellm_anthropic_wif_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "anthropic_wif",
"range": true,
"refId": "A"
},
{
"datasource": {
"type": "prometheus",
@ -5719,7 +5730,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_auth_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "auth",
"range": true,
"refId": "A"
"refId": "B"
},
{
"datasource": {
@ -5730,7 +5741,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_batch_write_to_db_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "batch_write_to_db",
"range": true,
"refId": "B"
"refId": "C"
},
{
"datasource": {
@ -5741,7 +5752,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_postgres_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "postgres",
"range": true,
"refId": "C"
"refId": "D"
},
{
"datasource": {
@ -5752,7 +5763,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_proxy_pre_call_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "proxy_pre_call",
"range": true,
"refId": "D"
"refId": "E"
},
{
"datasource": {
@ -5763,7 +5774,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis",
"range": true,
"refId": "E"
"refId": "F"
},
{
"datasource": {
@ -5774,7 +5785,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_org_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis_daily_org_spend_update_queue",
"range": true,
"refId": "F"
"refId": "G"
},
{
"datasource": {
@ -5785,7 +5796,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_tag_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis_daily_tag_spend_update_queue",
"range": true,
"refId": "G"
"refId": "H"
},
{
"datasource": {
@ -5796,7 +5807,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_team_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis_daily_team_spend_update_queue",
"range": true,
"refId": "H"
"refId": "I"
},
{
"datasource": {
@ -5807,7 +5818,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_window_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis_window_spend_update_queue",
"range": true,
"refId": "I"
"refId": "J"
},
{
"datasource": {
@ -5818,7 +5829,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_reset_budget_job_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "reset_budget_job",
"range": true,
"refId": "J"
"refId": "K"
},
{
"datasource": {
@ -5829,7 +5840,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_router_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "router",
"range": true,
"refId": "K"
"refId": "L"
},
{
"datasource": {
@ -5840,7 +5851,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_self_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "self",
"range": true,
"refId": "L"
"refId": "M"
}
],
"title": "Service latency p95 (litellm_<service>_latency)",
@ -5888,6 +5899,28 @@
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(rate(litellm_anthropic_wif_total_requests_total[$__rate_interval]))",
"legendFormat": "anthropic_wif",
"range": true,
"refId": "A"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(rate(litellm_anthropic_wif_cache_total_requests_total[$__rate_interval]))",
"legendFormat": "anthropic_wif_cache",
"range": true,
"refId": "B"
},
{
"datasource": {
"type": "prometheus",
@ -5897,7 +5930,7 @@
"expr": "sum(rate(litellm_auth_total_requests_total[$__rate_interval]))",
"legendFormat": "auth",
"range": true,
"refId": "A"
"refId": "C"
},
{
"datasource": {
@ -5908,7 +5941,7 @@
"expr": "sum(rate(litellm_batch_write_to_db_total_requests_total[$__rate_interval]))",
"legendFormat": "batch_write_to_db",
"range": true,
"refId": "B"
"refId": "D"
},
{
"datasource": {
@ -5919,7 +5952,7 @@
"expr": "sum(rate(litellm_postgres_total_requests_total[$__rate_interval]))",
"legendFormat": "postgres",
"range": true,
"refId": "C"
"refId": "E"
},
{
"datasource": {
@ -5930,7 +5963,7 @@
"expr": "sum(rate(litellm_proxy_pre_call_total_requests_total[$__rate_interval]))",
"legendFormat": "proxy_pre_call",
"range": true,
"refId": "D"
"refId": "F"
},
{
"datasource": {
@ -5941,7 +5974,7 @@
"expr": "sum(rate(litellm_redis_total_requests_total[$__rate_interval]))",
"legendFormat": "redis",
"range": true,
"refId": "E"
"refId": "G"
},
{
"datasource": {
@ -5952,7 +5985,7 @@
"expr": "sum(rate(litellm_redis_daily_org_spend_update_queue_total_requests_total[$__rate_interval]))",
"legendFormat": "redis_daily_org_spend_update_queue",
"range": true,
"refId": "F"
"refId": "H"
},
{
"datasource": {
@ -5963,7 +5996,7 @@
"expr": "sum(rate(litellm_redis_daily_tag_spend_update_queue_total_requests_total[$__rate_interval]))",
"legendFormat": "redis_daily_tag_spend_update_queue",
"range": true,
"refId": "G"
"refId": "I"
},
{
"datasource": {
@ -5974,7 +6007,7 @@
"expr": "sum(rate(litellm_redis_daily_team_spend_update_queue_total_requests_total[$__rate_interval]))",
"legendFormat": "redis_daily_team_spend_update_queue",
"range": true,
"refId": "H"
"refId": "J"
},
{
"datasource": {
@ -5985,7 +6018,7 @@
"expr": "sum(rate(litellm_redis_window_spend_update_queue_total_requests_total[$__rate_interval]))",
"legendFormat": "redis_window_spend_update_queue",
"range": true,
"refId": "I"
"refId": "K"
},
{
"datasource": {
@ -5996,7 +6029,7 @@
"expr": "sum(rate(litellm_reset_budget_job_total_requests_total[$__rate_interval]))",
"legendFormat": "reset_budget_job",
"range": true,
"refId": "J"
"refId": "L"
},
{
"datasource": {
@ -6007,7 +6040,7 @@
"expr": "sum(rate(litellm_router_total_requests_total[$__rate_interval]))",
"legendFormat": "router",
"range": true,
"refId": "K"
"refId": "M"
},
{
"datasource": {
@ -6018,7 +6051,7 @@
"expr": "sum(rate(litellm_self_total_requests_total[$__rate_interval]))",
"legendFormat": "self",
"range": true,
"refId": "L"
"refId": "N"
}
],
"title": "Service request rate (litellm_<service>_total_requests)",
@ -6066,6 +6099,28 @@
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(rate(litellm_anthropic_wif_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "anthropic_wif / {{error_class}}",
"range": true,
"refId": "A"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(rate(litellm_anthropic_wif_cache_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "anthropic_wif_cache / {{error_class}}",
"range": true,
"refId": "B"
},
{
"datasource": {
"type": "prometheus",
@ -6075,7 +6130,7 @@
"expr": "sum(rate(litellm_auth_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "auth / {{error_class}}",
"range": true,
"refId": "A"
"refId": "C"
},
{
"datasource": {
@ -6086,7 +6141,7 @@
"expr": "sum(rate(litellm_batch_write_to_db_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "batch_write_to_db / {{error_class}}",
"range": true,
"refId": "B"
"refId": "D"
},
{
"datasource": {
@ -6097,7 +6152,7 @@
"expr": "sum(rate(litellm_postgres_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "postgres / {{error_class}}",
"range": true,
"refId": "C"
"refId": "E"
},
{
"datasource": {
@ -6108,7 +6163,7 @@
"expr": "sum(rate(litellm_proxy_pre_call_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "proxy_pre_call / {{error_class}}",
"range": true,
"refId": "D"
"refId": "F"
},
{
"datasource": {
@ -6119,7 +6174,7 @@
"expr": "sum(rate(litellm_redis_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis / {{error_class}}",
"range": true,
"refId": "E"
"refId": "G"
},
{
"datasource": {
@ -6130,7 +6185,7 @@
"expr": "sum(rate(litellm_redis_daily_org_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis_daily_org_spend_update_queue / {{error_class}}",
"range": true,
"refId": "F"
"refId": "H"
},
{
"datasource": {
@ -6141,7 +6196,7 @@
"expr": "sum(rate(litellm_redis_daily_tag_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis_daily_tag_spend_update_queue / {{error_class}}",
"range": true,
"refId": "G"
"refId": "I"
},
{
"datasource": {
@ -6152,7 +6207,7 @@
"expr": "sum(rate(litellm_redis_daily_team_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis_daily_team_spend_update_queue / {{error_class}}",
"range": true,
"refId": "H"
"refId": "J"
},
{
"datasource": {
@ -6163,7 +6218,7 @@
"expr": "sum(rate(litellm_redis_window_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis_window_spend_update_queue / {{error_class}}",
"range": true,
"refId": "I"
"refId": "K"
},
{
"datasource": {
@ -6174,7 +6229,7 @@
"expr": "sum(rate(litellm_reset_budget_job_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "reset_budget_job / {{error_class}}",
"range": true,
"refId": "J"
"refId": "L"
},
{
"datasource": {
@ -6185,7 +6240,7 @@
"expr": "sum(rate(litellm_router_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "router / {{error_class}}",
"range": true,
"refId": "K"
"refId": "M"
},
{
"datasource": {
@ -6196,7 +6251,7 @@
"expr": "sum(rate(litellm_self_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "self / {{error_class}}",
"range": true,
"refId": "L"
"refId": "N"
}
],
"title": "Service failure rate (litellm_<service>_failed_requests)",

View file

@ -1,6 +1,6 @@
# LiteLLM All Prometheus Metrics dashboard
Every `litellm_*` metric family the proxy can expose on `/metrics` (136 families across 97 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
Every `litellm_*` metric family the proxy can expose on `/metrics` (141 families across 97 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
Import `grafana_dashboard.json` from **Dashboards > New > Import** and pick your Prometheus data source when prompted (the `DS_PROMETHEUS` variable). Counters are plotted as `rate()` over `$__rate_interval`, histograms as p50 / p95 / p99, gauges as the raw value grouped by the most useful label. Every query names the metric exactly as the proxy emits it (counters carry the `_total` suffix the Prometheus client adds), and `tests/unit/integrations/test_prometheus_metric_name_consistency.py` fails if a metric is renamed without updating this dashboard

View file

@ -1,7 +1,28 @@
FROM python:3.12-slim
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin
FROM $LITELLM_BUILD_IMAGE AS builder
COPY --from=uvbin /uv /usr/local/bin/uv
RUN apk add --no-cache python-3.13
ENV UV_PYTHON_DOWNLOADS=0 UV_LINK_MODE=copy
WORKDIR /app
RUN pip install --no-cache-dir httpx==0.28.1 pydantic==2.11.7
COPY litellm/proxy/lens/__init__.py litellm/proxy/lens/models.py litellm/proxy/lens/trace_store.py litellm/proxy/lens/analysis.py litellm/proxy/lens/worker.py /app/lens/
COPY deploy/lens/requirements.lock /tmp/requirements.lock
RUN uv venv --python python3.13 /app/.venv && \
uv pip sync --python /app/.venv/bin/python --require-hashes --only-binary :all: /tmp/requirements.lock
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
RUN : "${LITELLM_RELEASE_TAG:?Pass --build-arg LITELLM_RELEASE_TAG matching the gateway}"
RUN apk add --no-cache python-3.13
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG} \
PATH="/app/.venv/bin:${PATH}" \
PYTHONDONTWRITEBYTECODE=1
WORKDIR /app
COPY --from=builder /app/.venv /app/.venv
COPY litellm/proxy/lens/__init__.py litellm/proxy/lens/models.py litellm/proxy/lens/trace_store.py litellm/proxy/lens/analysis.py litellm/proxy/lens/worker.py litellm/proxy/lens/release.py /app/lens/
COPY litellm/proxy/lens/prompts/ /app/lens/prompts/
USER 65532:65532
CMD ["python", "-m", "lens.worker"]

View file

@ -1,4 +1,7 @@
**
!deploy/
!deploy/lens/
!deploy/lens/requirements.lock
!litellm/
!litellm/proxy/
!litellm/proxy/lens/
@ -7,5 +10,6 @@
!litellm/proxy/lens/trace_store.py
!litellm/proxy/lens/analysis.py
!litellm/proxy/lens/worker.py
!litellm/proxy/lens/release.py
!litellm/proxy/lens/prompts/
!litellm/proxy/lens/prompts/**

View file

@ -2,9 +2,74 @@
Lens reviews recorded activity and saves evidence-linked findings in the LiteLLM dashboard under Observability, Lens (`/ui/lens/`)
## Start a worker
## Install
Upgrade your existing LiteLLM proxy to a release that includes Lens with PostgreSQL and agent tracing. Configure one ClickHouse URL for trace writes, bounded reads, and Lens queries:
Build LiteLLM and its worker from the same source commit with the same release identity. The worker runs separately and connects to your gateway using a limited worker token
### New local installation
Install Docker with Compose and Git. This builds LiteLLM and its worker from the same checkout and starts the existing local tracing stack:
```bash
git clone https://github.com/BerriAI/litellm.git
cd litellm
export LITELLM_RELEASE_TAG="sha-$(git rev-parse HEAD)"
export LENS_WORKER_IMAGE="litellm-lens-worker:${LITELLM_RELEASE_TAG}"
export OPENAI_API_KEY='sk-...'
docker build --build-arg LITELLM_RELEASE_TAG="$LITELLM_RELEASE_TAG" \
-f deploy/lens/Dockerfile -t "$LENS_WORKER_IMAGE" .
docker compose -f docker/docker-compose.tracing.yml up -d --build
```
Open `http://localhost:4002/ui/` and sign in as `admin` with password `sk-1234`. Go to **Lens > Investigations > Connect worker**, choose a model and monthly budget, then **Get install command**. Expand **Using Docker Compose or Helm?** and copy the worker token. In the same terminal, run:
```bash
export LITELLM_URL=http://litellm:4000
export LENS_WORKER_TOKEN='<paste-your-worker-token>'
docker compose -f docker/docker-compose.tracing.yml -f deploy/lens/compose.yaml up -d
```
The worker joins the gateway's Docker network, and the dashboard shows **Worker connected**. Save the token privately for restarts and upgrades
This stack is for local evaluation: it binds to localhost and uses development database credentials. For a hosted deployment, keep your normal database, keys, networking, and deployment process. Build both images from one source revision with the same `LITELLM_RELEASE_TAG`, publish the worker to your registry, and set `LENS_WORKER_IMAGE` on LiteLLM to that image
### Existing LiteLLM installation
Keep your deployment and PostgreSQL database. A working gateway/worker pair can stay as it is until you upgrade both. For a gateway built from source, use its exact commit and `LITELLM_RELEASE_TAG`; a release version or the latest commit on `main` is not a substitute for that source identity
The public development package is `ghcr.io/berriai/litellm-lens-worker-dev:sha-<full-commit>`. It publishes amd64 images on Lens-related changes, so an arbitrary source commit may have no image. Check the exact image exists before using it. If it is unavailable, your gateway uses a different release identity, or you need native arm64, build the worker from the gateway's checkout:
```bash
export LITELLM_RELEASE_TAG='<gateway-release-identity>'
export LENS_WORKER_IMAGE='<your-registry>/litellm-lens-worker:<your-image-tag>'
docker build --build-arg LITELLM_RELEASE_TAG="$LITELLM_RELEASE_TAG" \
-f deploy/lens/Dockerfile -t "$LENS_WORKER_IMAGE" .
```
For a remote worker host, publish that image to a registry the host can pull from. Set the gateway's `LENS_WORKER_IMAGE` to the resulting image reference, restart the gateway using its normal deployment process, then copy its install command. Prefer the published image digest for hosted installations. Do not change the gateway's release identity just to accept another worker
For Kubernetes or Render, run the standalone worker using `LITELLM_URL` and `LENS_WORKER_TOKEN` from setup. Keep existing databases and secrets. The worker needs no inbound port.
## Helm
The componentized source chart at `helm/litellm` includes an optional Lens worker. Use the chart from the same checkout as your gateway and keep your component image overrides in your values. Configure PostgreSQL and ClickHouse as usual, install the chart, then obtain a limited worker token from Lens setup. Store it in a Kubernetes Secret and enable the worker in your values:
```yaml
lensWorker:
enabled: true
image:
repository: <your-worker-image-repository>
digest: sha256:<matching-worker-image-digest>
tokenSecret:
name: litellm-lens-worker
key: token
```
Set the worker repository and digest explicitly to an image built from the gateway's source commit and release identity. The chart connects the worker to the backend service. Keep these values and the Secret when upgrading the chart and update the gateway and worker image overrides together. `lensWorker.replicaCount` controls simultaneous investigations. To use a private registry or external proxy, set `lensWorker.image.repository`, `lensWorker.image.digest` (or `tag` for a source build), and `lensWorker.url`. A digest takes precedence over the tag. The dashboard uses the chart's worker image for standalone install commands too
## Standalone worker
Start with a source deployment that includes Lens, PostgreSQL, and agent tracing, and prepare its matching worker as described above. Configure one ClickHouse URL for trace writes, bounded reads, and Lens queries:
```yaml
general_settings:
@ -21,19 +86,19 @@ Retention changes require a proxy restart. ClickHouse removes expired rows durin
In **Lens > Investigations**, click **Connect worker**, choose an analysis model and monthly limit, then **Get install command**. Use **Advanced options** to select an existing virtual key or change the proxy URL if the server running Docker needs a different network address. Copy the command and run it on your server. The dashboard shows **Worker connected** when the container checks in
The command already contains the compatible worker image and one worker token. The selected virtual key stays on the proxy; its secret is never sent to the worker. No source checkout, environment file, or second LiteLLM deployment is needed. Keep the command private because it includes the token. The LiteLLM release provides the dashboard and APIs; the container only runs background analysis
The command already contains the compatible worker image and one worker token. The selected virtual key stays on the proxy; its secret is never sent to the worker. Once the matching image is available on the worker host, no second LiteLLM deployment is needed. Keep the command private because it includes the token. The LiteLLM release provides the dashboard and APIs; the container only runs background analysis
The dashboard and Compose file pin a verified worker image by digest. The image uses Linux amd64, and the generated command selects that platform. CI also publishes immutable `:sha-<commit>` tags for successful worker builds on `main`. Keep the worker image compatible with your gateway version
The dashboard uses the gateway's `LENS_WORKER_IMAGE` override when set. Public `:sha-<commit>` development images must match both the gateway commit and release identity. Build from source for the worker host's native architecture
After upgrading the gateway, update the worker image and redeploy it while keeping its proxy URL and token. Existing containers do not update automatically. If an investigation reports a worker compatibility error, update the image before retrying
For deployments managed with Compose, download `compose.yaml` and provide `LITELLM_URL` and `LENS_WORKER_TOKEN` in an environment file. Its default image is already selected:
For deployments managed with Compose, download `compose.yaml` and provide `LITELLM_URL`, `LENS_WORKER_TOKEN`, and an explicit `LENS_WORKER_IMAGE` in a private environment file:
```bash
docker compose --env-file /path/to/lens.env -f compose.yaml up -d
```
Developers can build locally with `LENS_WORKER_IMAGE=litellm-lens-worker:local docker compose -f deploy/lens/compose.yaml -f deploy/lens/compose.build.yaml up -d --build`
To work on Lens itself, `make lens-dev` runs the proxy, a worker from source and the hot-reload dashboard together; set `LENS_DEV_PROXY_PORT` / `LENS_DEV_UI_PORT` to move them off 4000/3000. For a local container build, set `LENS_WORKER_IMAGE=litellm-lens-worker:local` and `LITELLM_RELEASE_TAG` to the gateway's release tag, then use `docker compose -f deploy/lens/compose.yaml -f deploy/lens/compose.build.yaml up -d --build`
The generated command gives the worker 1 GiB of temporary memory-backed storage, shared across parallel reviews. Change `size=1g` in the Docker command or set `LENS_WORKER_TMP_SIZE` with Compose to fit your server and workload. A storage failure marks the scan as failed, cleans up temporary traces, and leaves the worker available for other scans; it does not silently truncate the review. Existing workers must be recreated with the new image and mount options
@ -107,6 +172,37 @@ curl "$LITELLM_URL/lens/$LENS_ID/runs/$BATCH_ID" -H "Authorization: Bearer $LITE
Creation queues the first batch. Posting to `/lens/{id}/runs` queues another, or returns the existing active batch. The run response contains its ID under `jobs[0].id`. Poll the batch URL for status, findings and assessments. List responses omit large result payloads; request a batch to retrieve them. Supply an optional complete `settings` object on the runs POST for a one-off override; the saved lens stays unchanged. Selection accepts `team_id`, exact `filters`, and opaque `execution_ids` returned by `/lens/preview/sample`. Preview accepts `offset` and `as_of` to keep the time window fixed while paging. Feedback uses `PATCH /lens/{id}/findings/{finding_id}` with `status` and `reason`
## Local development
`make lens-dev ARGS=--seed` starts the full dev stack. The live dashboard is at `http://localhost:3000/ui/lens/`, with login at `http://localhost:3000/ui/login/`. Next.js forwards API requests to the proxy on port 4000, so login and navigation stay in the live UI and edits hot-reload
The default is Next.js dev with no production build (`LENS_DEV_BUILD_UI=0`). Set `LENS_DEV_BUILD_UI=1` when you also want a fresh static dashboard at `http://localhost:4000/ui/`. Build output goes to `.lens-dev/logs/ui-build.log`; a failed build stops startup. Both modes keep the live dashboard on port 3000. Startup checks the live login route before seeding and fails with the UI log path if Next.js exits. `LENS_DEV_STARTUP_TIMEOUT_SECONDS` controls startup readiness retries (default 300; `LENS_DEV_READINESS_REQUEST_TIMEOUT_SECONDS` caps each HTTP probe, default 5)
For local fixture data, run `make lens-dev ARGS=--seed`. Use `make lens-dev ARGS="--seed large"` for 2,000 fixture copies spread over the last 24 hours, about 860,000 spans with linked request logs, plus three long sessions of roughly 1,150, 9,200 and 92,000 spans in a single trace for drawer paging and the oversized read path. Their trace IDs are printed at the end. To seed a running stack without restarting it, use `make lens-dev ARGS="--seed-only --seed large --copies 100"`. Every profile replays one copy of every checked-in capture through authenticated `/v1/traces`, including failures, retries, streaming and multiple agent frameworks, and verifies linked spend totals through the proxy. Large seeds then copy that first copy inside ClickHouse and PostgreSQL with `INSERT ... SELECT`, rewriting trace, span and call IDs so each copy keeps its own spend, and verify the last copy through the proxy
Seeds append fresh IDs on every invocation and spread copies over recent timestamps. Restarts without `SEED` do not add data. Lens excludes activity received in the last two minutes, so wait two minutes after seeding before checking investigation previews. `LENS_DEV_SEED_COPIES` overrides total copies. Large seeds test data volume and pagination, rather than concurrent ingestion throughput or review accuracy. They can use substantial disk space; adjust `--copies` for your machine. Seeding expects the generated local tracing configuration. The old `run_tracing_proxy_local.sh --seed` command forwards to Lens dev, using its ports and saved master key
Local ingestion limits are explicit and configurable. Set OTLP and ClickHouse variables before starting the proxy and seeder so both processes use the same settings. Invalid, zero and negative values fail instead of silently falling back. Changing these limits does not require rebuilding Rust
| Environment variable | Default | Controls |
| --- | --- | --- |
| `LENS_DEV_SEED_COPIES` | 1 default, 2000 large | Total fixture copies |
| `LENS_DEV_SEED_TIMEOUT_SECONDS` | 120 | Seeder HTTP timeout |
| `OTLP_MAX_BODY_BYTES` | 16777216 | HTTP body and decompressed payload bytes |
| `OTLP_MAX_CONCURRENT_INGESTS` | 2 | Concurrent proxy ingestion requests |
| `OTLP_MAX_ATTRIBUTE_VALUE_BYTES` | 65536 | Stored attribute/content bytes |
| `OTLP_MAX_DECODE_DEPTH` | 32 | Nested decode depth |
| `OTLP_MAX_DECODE_NODES` | 65536 | JSON values or protobuf fields per export |
| `OTLP_MAX_SPANS` | 4096 | Spans per export |
| `OTLP_MAX_ATTRIBUTES` | 256 | Attributes per resource, scope, span, event or link |
| `OTLP_MAX_EVENTS` | 256 | Events per span |
| `OTLP_MAX_LINKS` | 256 | Links per span |
| `OTLP_MAX_DECODED_SPAN_BYTES` | 16777216 | Decoded span allocation budget |
| `CLICKHOUSE_TRACE_MAX_INSERT_BYTES` | 67108864 | Encoded trace or spend insert bytes |
| `CLICKHOUSE_INSERT_TIMEOUT_SECONDS` | 30 | ClickHouse insert HTTP timeout |
The wire parsers also enforce their library recursion limits (128 levels for JSON, 100 for protobuf). Raising the configured depth does not remove those parser limits.
## Quality evaluation
Run the checked-in cases against a configured real model. Expected labels are used only for scoring, never passed to the model. Dev and held-out cases include missing outcomes, failed tools, recovery, handoffs, unsupported claims, repeated work, long evidence and prompt injection. The background option adds clean arithmetic traces to test rare-issue discovery at scale; those repeated synthetic cases do not establish accuracy on every production workload
@ -130,3 +226,19 @@ The Lens API now uses `/lens` instead of `/engine`, list responses use `lenses`,
Stop workers and let active scans finish before upgrading. Deploy proxy instances together: older proxies cannot use the renamed database tables. The schema migration renames the three Lens tables and the run-history identifier column in place, preserving saved investigations, findings, history, worker credentials, and billing assignments. Existing migration files retain their original names and checksums
Upgrades using `--use_prisma_db_push` stop before schema changes if any legacy Lens table exists, preventing Prisma from dropping saved data. Apply `litellm-proxy-extras/litellm_proxy_extras/migrations/20261001100000_rename_lens/migration.sql` to the configured database schema before retrying. Deployments already using migration history can instead start without `--use_prisma_db_push` to apply the shipped migration normally. Fresh databases and databases already using the renamed tables can continue using database push
## Release compatibility
Gateway and worker builds carry the same `LITELLM_RELEASE_TAG`. A worker announces its release and protocol before claiming an investigation. A mismatch returns HTTP 409 with the required image, leaving queued investigations untouched. During a rolling upgrade, workers wait for a gateway from their release
The dashboard reads its image from the running gateway. `LENS_WORKER_IMAGE` overrides the registry/image for private deployments. Set an explicit `LENS_WORKER_IMAGE` for worker-only Compose. Verify that the image exists and matches the gateway before deploying it
For source development, use `make lens-dev`, which gives the proxy and source worker the same commit identity. For custom containers, build both from the same checkout with `--build-arg LITELLM_RELEASE_TAG=sha-$(git rev-parse HEAD)` and set the proxy's `LENS_WORKER_IMAGE` to the worker image you built. An unlabelled custom build refuses worker setup and claims instead of guessing from the Python package version. Normal package-index installations use their installed release version
The hourly development pipeline pins all component images to the same selected commit and publishes its chart only after every build and worker smoke test succeeds. The public commit-tagged worker workflow publishes to `ghcr.io/berriai/litellm-lens-worker-dev` on Lens-related changes, so an arbitrary `main` commit may require building your own pair; do not substitute the newest available worker
## Worker dependencies
The worker uses the same digest-pinned Wolfi base and Python version as the component images. Python dependencies and their hashes are locked in `deploy/lens/requirements.lock`. To update them, edit `deploy/lens/requirements.in`, then run `uv pip compile --universal --python-version 3.13 --generate-hashes --no-emit-index-url deploy/lens/requirements.in -o deploy/lens/requirements.lock`. The image installs only the locked wheels with hash verification. CI builds and scans both native architectures

View file

@ -3,4 +3,6 @@ services:
build:
context: ../..
dockerfile: deploy/lens/Dockerfile
args:
LITELLM_RELEASE_TAG: ${LITELLM_RELEASE_TAG:?Set the release tag used by the gateway}
image: litellm-lens-worker:local

View file

@ -1,6 +1,6 @@
services:
lens-worker:
image: ${LENS_WORKER_IMAGE:-ghcr.io/berriai/litellm-lens-worker@sha256:44f0597c7583dcfef999ece9a8bc02cfeb9f0f5167a1221cee3bd10b1b79271b}
image: ${LENS_WORKER_IMAGE:-${LITELLM_VERSION:+ghcr.io/berriai/litellm-lens-worker:v}${LITELLM_VERSION:-}}
environment:
LITELLM_URL: ${LITELLM_URL:?Set the URL reachable from this container}
LENS_WORKER_TOKEN: ${LENS_WORKER_TOKEN:?Create a worker credential in the Lens UI}

7
deploy/lens/config.yaml Normal file
View file

@ -0,0 +1,7 @@
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
tracing:
store:
type: clickhouse
url: os.environ/CLICKHOUSE_URL
retention_days: 14

View file

@ -0,0 +1,2 @@
httpx==0.28.1
pydantic==2.13.4

View file

@ -0,0 +1,172 @@
# This file was autogenerated by uv via the following command:
# uv pip compile --universal --python-version 3.13 --generate-hashes --no-emit-index-url deploy/lens/requirements.in -o deploy/lens/requirements.lock
annotated-types==0.8.0 \
--hash=sha256:13b2beaad985e05e2d6407ee4c4f35590b11f8d693a258a561055cac8f64cab7 \
--hash=sha256:f072f4d804ea359e4eaf198b1af7a8b0943881a87f31bb764f8bf219bb9419e0
# via pydantic
anyio==4.15.1 \
--hash=sha256:6152fdbbf9a77fdec97731721bebf7c4c44f7c29b424b0065826173efc7ed101 \
--hash=sha256:9f28306018cbd6d329e64a36d58256edff76dd996fe423bc957326e578b82a94
# via httpx
certifi==2026.7.22 \
--hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 \
--hash=sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55
# via
# httpcore
# httpx
h11==0.16.0 \
--hash=sha256:4e35b956cf45792e4caa5885e69fba00bdbc6ffafbfa020300e549b208ee5ff1 \
--hash=sha256:63cf8bbe7522de3bf65932fda1d9c2772064ffb3dae62d55932da54b31cb6c86
# via httpcore
httpcore==1.0.9 \
--hash=sha256:2d400746a40668fc9dec9810239072b40b4484b640a8c38fd654a024c7a1bf55 \
--hash=sha256:6e34463af53fd2ab5d807f399a9b45ea31c3dfa2276f15a2c3f00afff6e176e8
# via httpx
httpx==0.28.1 \
--hash=sha256:75e98c5f16b0f35b567856f597f06ff2270a374470a5c2392242528e3e3e42fc \
--hash=sha256:d909fcccc110f8c7faf814ca82a9a4d816bc5a6dbfea25d6591d6985b8ba59ad
# via -r deploy/lens/requirements.in
idna==3.20 \
--hash=sha256:a7db850025b95ded1eae8a46181a1a6c56c92c96f0e2b005d9ff8dc0210cab44 \
--hash=sha256:ab7ae7122974553370f0bdb919e1a960b2cd1bc1ef0276416d896db81c14582c
# via
# anyio
# httpx
pydantic==2.13.4 \
--hash=sha256:45a282cde31d808236fd7ea9d919b128653c8b38b393d1c4ab335c62924d9aba \
--hash=sha256:c40756b57adaa8b1efeeced5c196f3f3b7c435f90e84ea7f443901bec8099ef6
# via -r deploy/lens/requirements.in
pydantic-core==2.46.4 \
--hash=sha256:00c603d540afdd6b80eb39f078f33ebd46211f02f33e34a32d9f053bba711de0 \
--hash=sha256:0186750b482eefa11d7f435892b09c5c606193ef3375bcf94aa00ae6bfb66262 \
--hash=sha256:041bde0a48fd37cf71cab1c9d56d3e8625a3793fef1f7dd232b3ff37e978ecda \
--hash=sha256:0c563b08bca408dc7f65f700633d8442fffb2421fc47b8101377e9fd65051ff0 \
--hash=sha256:0cbe8b01f948de4286c74cdd6c667aceb38f5c1e26f0693b3983d9d74887c65e \
--hash=sha256:0ce40cd7b21210e99342afafbd4d0f76d784eb5b1d60f3bdc566be4983c6c73b \
--hash=sha256:0e96592440881c74a213e5ad528e2b24d3d4f940de2766bed9010ab1d9e51594 \
--hash=sha256:10e17cbb10a330363733efc4d7c4d0dd827ac0909b8f6a6542298fed1ea62f29 \
--hash=sha256:133878133d271ade3d41d1bfb2a45ec38dbdbda40bc065921c6b04e4630127e2 \
--hash=sha256:14d4edf427bdcf950a8a02d7cb44a08614388dd6e1bdcbf4f67504fa7887da9c \
--hash=sha256:14f4c5d6db102bd796a627bbb3a17b4cf4574b9ae861d8b7c9a9661c6dd3362d \
--hash=sha256:17299feefe090f2caa5b8e37222bb5f663e4935a8bfa6931d4102e5df1a9f398 \
--hash=sha256:184c081504d17f1c1066e430e117142b2c77d9448a97f7b65c6ac9fd9aee238d \
--hash=sha256:18e5ceec2ab67e6d5f1a9085e5a24c9c4e2ac4545730bfe668680bca05e555f3 \
--hash=sha256:19e51f073cd3df251856a8a4189fbdf1de4012c3ebacfb1884f94f1eb406079f \
--hash=sha256:1a7dd0b3ee80d90150e3495a3a13ac34dbcbfd4f012996a6a1d8900e91b5c0fb \
--hash=sha256:1d8ba486450b14f3b1d63bc521d410ec7565e52f887b9fb671791886436a42f7 \
--hash=sha256:2108ba5c1c1eca18030634489dc544844144ee36357f2f9f780b93e7ddbb44b5 \
--hash=sha256:228ee9bae8bef5b1e97ec58302f80357c37199e0d0a99174e138d28e6957b9d9 \
--hash=sha256:23ace664830ee0bfe014a0c7bc248b1f7f25ed7ad103852c317624a1083af462 \
--hash=sha256:2412e734dcb48da14d4e4006b82b46b74f2518b8a26ee7e58c6844a6cd6d03c4 \
--hash=sha256:29c61fc04a3d840155ff08e475a04809278972fe6aef51e2720554e96367e34b \
--hash=sha256:2f84c03c8607173d16b5a854ec68a2f9079ae03237a54fb506d13af47e1d018d \
--hash=sha256:3009f12e4e90b7f88b4f9adb1b0c4a3d58fe7820f3238c190047209d148026df \
--hash=sha256:3245406455a5d98187ec35530fd772b1d799b26667980872c8d4614991e2c4a2 \
--hash=sha256:3447661d99f75a3683a4cf5c87da72f2161964611864dbbeac7fbb118bb4bfc0 \
--hash=sha256:372429a130e469c9cd698925ce5fc50940b7a1336b0d82038e63d5bbc4edc519 \
--hash=sha256:395aebd9183f9d112f569aeb5b2214d1a10a33bec8456447f7fbdfa51d38d4cd \
--hash=sha256:3a233125ac121aa3ffba9a2b59edfc4a985a76092dc8279586ab4b71390875e7 \
--hash=sha256:3be77f45df024d789a672ae34f8b06fb346c4f9f46ea714956660ea4862e89ac \
--hash=sha256:3bf92c5d0e00fefaab325a4d27828fe6b6e2a21848686b5b60d2d9eeb09d76c6 \
--hash=sha256:3ecbc122d18468d06ca279dc26a8c2e2d5acb10943bb35e36ae92096dc3b5565 \
--hash=sha256:3fb702cd90b0446a3a1c5e470bfa0dd23c0233b676a9099ddcc964fa6ca13898 \
--hash=sha256:428e04521a40150c85216fc8b85e8d39fece235a9cf5e383761238c7fa9b96fb \
--hash=sha256:432c179df7874eeb73307aad2df0755e1ae0efa61ff0ea89b93e194411ae3928 \
--hash=sha256:4a05d69cba51d852c5c3e92758653245a50c0b646ced0cf05bd793ed592839d6 \
--hash=sha256:4c63ebc82684aa89d9a3bcbd13d515b3be44250dc68dd3bd81526c1cb31286c3 \
--hash=sha256:4fc73cb559bdb54b1134a706a2802a4cddd27a0633f5abb7e53056268751ac6a \
--hash=sha256:4fcbe087dbc2068af7eda3aa87634eba216dbda64d1ae73c8684b621d33f6596 \
--hash=sha256:56cb4851bcaf3d117eddcef4fe66afd750a50274b0da8e22be256d10e5611987 \
--hash=sha256:5855698a4856556d86e8e6cd8434bc3ac0314ee8e12089ae0e143f64c6256e4e \
--hash=sha256:5a4330cdbc57162e4b3aa303f588ba752257694c9c9be3e7ebb11b4aca659b5d \
--hash=sha256:5b712b53160b79a5850310b912a5ef8e57e56947c8ad690c227f5c9d7e561712 \
--hash=sha256:5d5902252db0d3cedf8d4a1bc68f70eeb430f7e4c7104c8c476753519b423008 \
--hash=sha256:617d7e2ca7dcb8c5cf6bcb8c59b8832c94b36196bbf1cbd1bfb56ed341905edd \
--hash=sha256:62f875393d7f270851f20523dd2e29f082bcc82292d66db2b64ea71f64b6e1c1 \
--hash=sha256:633147d34cf4550417f12e2b1a0383973bdf5cdfde212cb09e9a581cf10820be \
--hash=sha256:66ce7632c22d837c95301830e111ad0128a32b8207533b60896a96c4915192ea \
--hash=sha256:6b3ace8194b0e5204818c92802dcdca7fc6d88aabbb799d7c795540d9cd6d292 \
--hash=sha256:6f2eeda33a839975441c86a4119e1383c50b47faf0cbb5176985565c6bb02c33 \
--hash=sha256:7027560ee92211647d0d34e3f7cd6f50da56399d26a9c8ad0da286d3869a53f3 \
--hash=sha256:7283d57845ecf5a163403eb0702dfc220cc4fbdd18919cb5ccea4f95ee1cdab4 \
--hash=sha256:7a5f930472650a82629163023e630d160863fce524c616f4e5186e5de9d9a49b \
--hash=sha256:7bfb192b3f4b9e8a89b6277b6ce787564f62cfd272055f6e685726b111dc7826 \
--hash=sha256:811ff8e9c313ab425368bcbb36e5c4ebd7108c2bbf4e4089cfbb0b01eff63fac \
--hash=sha256:8233f2947cf85404441fd7e0085f53b10c93e0ee78611099b5c7237e36aacbf7 \
--hash=sha256:82cf5301172168103724d49a1444d3378cb20cdee30b116a1bd6031236298a5d \
--hash=sha256:8358a950c8909158e3df31538a7e4edc2d7265a7c54b47f0864d9e5bae9dcebf \
--hash=sha256:85bb3611ff1802f3ee7fdd7dbff26b56f343fb432d57a4728fdd49b6ef35e2f4 \
--hash=sha256:86e1a4418c6cd97d60c95c71164158eaf7324fae7b0923264016baa993eba6fc \
--hash=sha256:8b9bab013d1c7a79d3501ff86d0bc9c31bf587db4551677b96bec07df78c6b15 \
--hash=sha256:8c5dac79fa1614d1e06ca695109c6105923bd9c7d1d6c918d4e637b7e6b32fd3 \
--hash=sha256:8d0820e8192167f80d88d64038e609c31452eeca865b4e1d9950a27a4609b00b \
--hash=sha256:8daafc69c93ee8a0204506a3b6b30f586ef54028f52aeeeb5c4cfc5184fd5914 \
--hash=sha256:9037063db01f09b09e237c282b6792bd4da634b5402c4e7f0c61effed7701a04 \
--hash=sha256:905a0ed8ea6f2d61c1738835f99b699348d7857379083e5fc497fa0c967a407c \
--hash=sha256:90884113d8b48f760e9587002789ddd741e76ab9f89518cd1e43b1f1a52ec44b \
--hash=sha256:91a06d2e259ecfbd8c901d70c3c507900458498142b3026a296b7de4d1322cc9 \
--hash=sha256:926c9541b14b12b1681dca8a0b75feb510b06c6341b70a8e500c2fdcff837cce \
--hash=sha256:9401557acd873c3a7f3eb9383edef8ac4968f9510e340f4808d427e75667e7b4 \
--hash=sha256:9551187363ffc0de2a00b2e47c25aeaeb1020b69b668762966df15fc5659dd5a \
--hash=sha256:962ccbab7b642487b1d8b7df90ef677e03134cf1fd8880bf698649b22a69371f \
--hash=sha256:97e7cf2be5c77b7d1a9713a05605d49460d02c6078d38d8bef3cbe323c548424 \
--hash=sha256:9aa768456404a8bf48a4406685ac2bec8e72b62c69313734fa3b73cf33b3a894 \
--hash=sha256:9bc519fbf2b7578398853d815009ae5e4d4603d12f4e3f91da8c06852d3da3e9 \
--hash=sha256:9d56801be94b86a9da183e5f3766e6310752b99ff647e38b09a9500d88e46e76 \
--hash=sha256:9f444c499b3eefd3a92e348059471ea0c3a6e303d9c1cec09fa748fd9f895201 \
--hash=sha256:9fa8ae11da9e2b3126c6426f147e0fba88d96d65921799bb30c6abd1cb2c97fb \
--hash=sha256:a0f62d0a58f4e7da165457e995725421e0064f2255d8eccebc49f41bbc23b109 \
--hash=sha256:a396dcc17e5a0b164dbe026896245a4fa9ff402edca1dff0be3d53a517f74de4 \
--hash=sha256:aaa2a54443eff1950ba5ddc6b6ccda0d9c84a364276a62f969bdf2a390650848 \
--hash=sha256:ad785e92e6dc634c21555edc8bd6b64957ab844541bcb96a1366c202951ae526 \
--hash=sha256:af8244b2bef6aaad6d92cda81372de7f8c8d36c9f0c3ea36e827c60e7d9467a0 \
--hash=sha256:b078afbc25f3a1436c7a1d2cd3e322497ee99615ba97c563566fdf46aff1ee01 \
--hash=sha256:b2f69dec1725e79a012d920df1707de5caf7ed5e08f3be4435e25803efc47458 \
--hash=sha256:b8458003118a712e66286df6a707db01c52c0f52f7db8e4a38f0da1d3b94fc4e \
--hash=sha256:bb63e0198ca18aad131c089b9204c23079c3afa95487e561f4c522d519e55aba \
--hash=sha256:bfec22eab3c8cc2ceec0248aec886624116dc079afa027ecc8ad4a7e62010f8a \
--hash=sha256:c1747f85cee84c26985853c6f3d9bd3e75da5212912443fa111c113b9c246f39 \
--hash=sha256:c1b3f518abeca3aa13c712fd202306e145abf59a18b094a6bafb2d2bbf59192c \
--hash=sha256:c50f2528cf200c5eed56faf3f4e22fcd5f38c157a8b78576e6ba3168ec35f000 \
--hash=sha256:c68fcd102d71ea85c5b2dfac3f4f8476eff42a9e078fd5faefff6d145063536b \
--hash=sha256:c7a7bd4e39e8e4c12c39cd480356842b6a8a06e41b23a55a5e3e191718838ddf \
--hash=sha256:c94f0688e7b8d0a67abf40e57a7eaaecd17cc9586706a31b76c031f63df052b4 \
--hash=sha256:cbaf13819775b7f769bf4a1f066cb6df7a28d4480081a589828ef190226881cd \
--hash=sha256:cd2213145bcc2ba85884d0ac63d222fece9209678f77b9b4d76f054c561adb28 \
--hash=sha256:ce5c1d2a8b27468f433ca974829c44060b8097eedc39933e3c206a90ee49c4a9 \
--hash=sha256:d396ec2b979760aaf3218e76c24e65bd0aca24983298653b3a9d7a45f9e47b30 \
--hash=sha256:d51026d73fcfd93610abc7b27789c26b313920fcfb20e27462d74a7f8b06e983 \
--hash=sha256:d80ee3d731373b24cebbc10d689ca4ee1875caf0d5703a245db18efd4dd37fc1 \
--hash=sha256:d995260fdf4e1db774581b4900e0f832abe3c7c84996726bbc161b19c8f29e76 \
--hash=sha256:da4b951fe36dc7c3a1ccb4e3cd1747c3542b8c9ceede8fc86cae054e764485f5 \
--hash=sha256:daa27d92c36f24388fe3ad306b174781c747627f134452e4f128ea00ce1fe8c4 \
--hash=sha256:db06ffe51636ffe9ca531fe9023dd64bdd794be8754cb5df57c5498ae5b518a7 \
--hash=sha256:e0d65b8c354be7fb5f720c3caa8bc940bc2d20ce749c8e06135f07f8ed95dd7c \
--hash=sha256:e68b7a074f65a2fd746c52a7ce6142ab7006074ac269ace0c25cd8ba171f8066 \
--hash=sha256:e739fee756ba1010f8bcccb534252e85a35fe45ae92c295a06059ce58b74ccd3 \
--hash=sha256:e846ae7835bf0703ae43f534ab79a867146dadd59dc9ca5c8b53d5c8f7c9ef02 \
--hash=sha256:e9c26f834c65f5752f3f06cb08cb86a913ceb7274d0db6e267808a708b46bc89 \
--hash=sha256:ea793e075b70290d89d8142074262885d3f7da19634845135751bd6344f73b50 \
--hash=sha256:f027324c56cd5406ca49c124b0db10e56c69064fec039acc571c29020cc87c76 \
--hash=sha256:f13a646d65d09fbf1bc6b3a9635d30095c8e7e5cc419ff35ecc563c5fd04cd49 \
--hash=sha256:f47286a97f0bc9b8859519809077b91b2cefe4ae47fcbf5e466a009c1c5d742b \
--hash=sha256:f747929cf940cddb5b3668a390056ddd5ba2e5010615ea2dcf4f9c4f3ab8791d \
--hash=sha256:f99626688942fb746e545232e7726926f3be91b5975f8b55327665fafda991c7 \
--hash=sha256:f9fa868638bf362d3d138ea55829cefb3d5f4b0d7f142234382a15e2485dbec4 \
--hash=sha256:fbdb89b3e1c94a30cc5edfce477c6e6a5dc4d8f84665b455c27582f211a1c72c \
--hash=sha256:fc010ab034c8c7452522748bf937df58020d256ccae0874463d1f4d01758af8e \
--hash=sha256:fc3e9034a63de20e15e8ade85358bc6efc614008cab72898b4b4952bea0509ff \
--hash=sha256:fd8b3d9fd264be37976686c7f65cd52a83f5e84f4bfd2adf9c1d469676bbb6ae
# via pydantic
typing-extensions==4.16.0 \
--hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 \
--hash=sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5
# via
# anyio
# pydantic
# pydantic-core
# typing-inspection
typing-inspection==0.4.4 \
--hash=sha256:547274fa6b0a561ccf549cc9524b999a578e737d015d8709d021f9d0d13bea47 \
--hash=sha256:65b8397ba37ccbce054456aaccddfc91e6e3083c92824df348d96ca832f3f147
# via pydantic

91
deploy/lens/stack.yaml Normal file
View file

@ -0,0 +1,91 @@
name: litellm-lens
services:
litellm:
image: ghcr.io/berriai/litellm:${LITELLM_VERSION:?Set LITELLM_VERSION to a published release, without the v prefix}
entrypoint:
- python3
- -c
- |
import os, sys
from urllib.parse import quote
postgres_password = quote(os.environ["POSTGRES_PASSWORD"], safe="")
clickhouse_password = quote(os.environ["CLICKHOUSE_PASSWORD"], safe="")
os.environ["DATABASE_URL"] = f"postgresql://litellm:{postgres_password}@db:5432/litellm"
os.environ["CLICKHOUSE_URL"] = f"http://default:{clickhouse_password}@clickhouse:8123"
os.execv("docker/prod_entrypoint.sh", ["docker/prod_entrypoint.sh", *sys.argv[1:]])
command: ["--config", "/app/lens-config.yaml", "--port", "4000"]
environment:
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?Set a strong master key}
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?Set a permanent encryption key and keep it across upgrades}
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:?Set a permanent database password}
STORE_MODEL_IN_DB: "True"
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:?Set a permanent ClickHouse password}
LENS_WORKER_IMAGE: ghcr.io/berriai/litellm-lens-worker:v${LITELLM_VERSION}
volumes:
- ./config.yaml:/app/lens-config.yaml:ro
ports:
- "127.0.0.1:${LITELLM_PORT:-4000}:4000"
networks: [proxy, storage]
depends_on:
db:
condition: service_healthy
clickhouse:
condition: service_healthy
restart: unless-stopped
lens-worker:
profiles: [lens]
image: ghcr.io/berriai/litellm-lens-worker:v${LITELLM_VERSION}
environment:
LITELLM_URL: http://litellm:4000
LENS_WORKER_TOKEN: ${LENS_WORKER_TOKEN:-}
depends_on: [litellm]
networks: [proxy]
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:rw,noexec,nosuid,size=${LENS_WORKER_TMP_SIZE:-1g}
cap_drop: [ALL]
security_opt: [no-new-privileges:true]
db:
image: postgres:16
environment:
POSTGRES_DB: litellm
POSTGRES_USER: litellm
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD}
networks: [storage]
volumes:
- postgres_data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U litellm -d litellm"]
interval: 5s
timeout: 5s
retries: 20
restart: unless-stopped
clickhouse:
image: clickhouse/clickhouse-server:26.9.6.6
environment:
CLICKHOUSE_USER: default
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD}
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: "1"
volumes:
- clickhouse_data:/var/lib/clickhouse
healthcheck:
test: ["CMD", "clickhouse-client", "--user", "default", "--password", "${CLICKHOUSE_PASSWORD}", "--query", "SELECT 1"]
interval: 5s
timeout: 5s
retries: 20
restart: unless-stopped
networks: [storage]
networks:
proxy:
storage:
internal: true
volumes:
postgres_data:
clickhouse_data:

View file

@ -0,0 +1,44 @@
services:
litellm:
image: ${LITELLM_IMAGE:?Set the native-enabled gateway image}
environment:
LITELLM_ADMIN_AGENT_URL: http://liteadmin:10000
ADMIN_AGENT_SERVICE_TOKEN: ${ADMIN_AGENT_SERVICE_TOKEN:?Set a shared worker token}
PROXY_BASE_URL: ${LITELLM_PUBLIC_URL:?Set the existing HTTPS gateway URL}
liteadmin:
image: ${LITELLM_IMAGE:?Set the same native-enabled image used by the gateway}
command: ["--admin-agent"]
restart: unless-stopped
init: true
read_only: true
cap_drop: [ALL]
security_opt: [no-new-privileges:true]
stop_grace_period: 75s
environment:
CONNECTION_AUTH_MODE: native
LITELLM_BASE_URL: ${LITELLM_PUBLIC_URL:?Set the existing HTTPS gateway URL}
LITELLM_MODEL: ${LITELLM_ADMIN_MODEL:?Set a gateway model with tool support}
SLACK_BOT_TOKEN: ${SLACK_BOT_TOKEN:?Install the Slack app}
SLACK_APP_TOKEN: ${SLACK_APP_TOKEN:?Enable Socket Mode}
SLACK_WORKSPACE_ID: ${SLACK_WORKSPACE_ID:?Set the Slack workspace ID}
ADMIN_AGENT_SERVICE_TOKEN: ${ADMIN_AGENT_SERVICE_TOKEN:?Set a shared worker token}
CREDENTIAL_ENCRYPTION_KEY: ${CREDENTIAL_ENCRYPTION_KEY:?Set a persistent Fernet key}
STATE_DB: /var/data/events.sqlite3
ADMIN_READ_ONLY: ${ADMIN_READ_ONLY:-false}
OPENAI_AGENTS_DISABLE_TRACING: "1"
volumes:
- liteadmin_state:/var/data
tmpfs:
- /tmp:rw,noexec,nosuid,size=64m
healthcheck:
test: ["CMD", "/opt/liteadmin/bin/python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:10000/readyz', timeout=3)"]
interval: 30s
timeout: 5s
start_period: 30s
depends_on:
litellm:
condition: service_healthy
volumes:
liteadmin_state:

View file

@ -113,6 +113,8 @@ RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG}
USER root

View file

@ -122,6 +122,8 @@ RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG}
WORKDIR /app
USER root

View file

@ -5,6 +5,8 @@ services:
build:
context: ..
target: runtime
args:
LITELLM_RELEASE_TAG: ${LITELLM_RELEASE_TAG:-}
command: ["--config", "/app/tracing-config.yaml", "--port", "4000"]
environment:
LITELLM_MASTER_KEY: sk-1234
@ -15,6 +17,7 @@ services:
CLICKHOUSE_URL: http://default:local-tracing@clickhouse:8123
CLICKHOUSE_DATABASE: litellm
OPENAI_API_KEY: ${OPENAI_API_KEY:-}
LENS_WORKER_IMAGE: ${LENS_WORKER_IMAGE:-}
volumes:
- ./tracing-config.yaml:/app/tracing-config.yaml:ro
ports:

View file

@ -1,5 +1,11 @@
#!/bin/sh
if [ "$1" = "--admin-agent" ]; then
shift
export CONNECTION_AUTH_MODE=native
exec /opt/liteadmin/bin/litellm-admin-agent --web "$@"
fi
case "$USE_DDTRACE" in
[Tt][Rr][Uu][Ee])
export DD_TRACE_OPENAI_ENABLED="False"

View file

@ -6,6 +6,7 @@ from litellm_enterprise.enterprise_callbacks.send_emails.endpoints import (
from . import ui_crud_endpoints # side-effect: registers extra UI settings
from .audit_logging_endpoints import router as audit_logging_router
from .liteadmin import router as liteadmin_router
from .management_endpoints import management_endpoints_router
from .utils import _should_block_robots
@ -14,6 +15,7 @@ __all__ = ["router", "ui_crud_endpoints"]
router = APIRouter()
router.include_router(email_events_router)
router.include_router(audit_logging_router)
router.include_router(liteadmin_router)
router.include_router(management_endpoints_router)

View file

@ -0,0 +1,283 @@
from __future__ import annotations
import hashlib
import hmac
import html
import os
import re
import secrets
from collections.abc import Awaitable, Callable
from dataclasses import dataclass
from datetime import datetime, timedelta, timezone
from typing import Annotated, Final
from urllib.parse import urlencode, urlsplit
import httpx
from fastapi import APIRouter, Depends, HTTPException, Request
from fastapi.responses import HTMLResponse, RedirectResponse, Response
from pydantic import BaseModel, ConfigDict, Field, SecretStr, TypeAdapter, ValidationError
from litellm.llms.custom_httpx.http_handler import get_async_httpx_client
from litellm.proxy._experimental.mcp_server.oauth_utils import get_request_base_url
from litellm.proxy._types import LiteLLM_UserTable, LitellmUserRoles, UserAPIKeyAuth
from litellm.types.proxy.auth.auth_checks import UserNotFoundError
router: Final = APIRouter()
_PREFIX: Final = "/liteadmin/slack/connect/"
_COOKIE: Final = "__Host-litellm-slack-connect-"
_HEADERS: Final = {
"Cache-Control": "no-store",
"Referrer-Policy": "same-origin",
"X-Frame-Options": "DENY",
"X-Content-Type-Options": "nosniff",
"Content-Security-Policy": "default-src 'none'; style-src 'unsafe-inline'; form-action 'self'; frame-ancestors 'none'; base-uri 'none'",
}
class LinkDetails(BaseModel):
model_config = ConfigDict(frozen=True, strict=True, extra="forbid")
workspace_id: str = Field(min_length=1, max_length=64)
slack_user_id: str = Field(min_length=1, max_length=64)
email: str = Field(min_length=1, max_length=320)
class AdminSession(BaseModel):
model_config = ConfigDict(frozen=True)
user_id: str
credential: SecretStr
expires_at: float
@dataclass(frozen=True, slots=True)
class NativeAdminContext:
worker_url: str
service_token: SecretStr
client: httpx.AsyncClient
session_user: Callable[[Request], Awaitable[str | None]]
load_user: Callable[[str], Awaitable[LiteLLM_UserTable | None]]
mint_session: Callable[[LiteLLM_UserTable], AdminSession]
async def worker_request(self, token: str, session: AdminSession | None = None) -> httpx.Response:
if re.fullmatch(r"[A-Za-z0-9_-]{43}", token) is None:
raise HTTPException(410, "Connection link expired. Send connect in Slack for a new link")
try:
response: Final = await self.client.request(
"GET" if session is None else "POST",
f"{self.worker_url}/internal/liteadmin/links/{token}",
headers={"X-LiteLLM-Admin-Agent-Token": self.service_token.get_secret_value()},
json=None
if session is None
else {
"user_id": session.user_id,
"credential": session.credential.get_secret_value(),
"expires_at": session.expires_at,
},
timeout=15,
follow_redirects=False,
)
except httpx.HTTPError:
raise HTTPException(503, "LiteAdmin is temporarily unavailable") from None
if response.status_code == 410:
raise HTTPException(410, "Connection link expired. Send connect in Slack for a new link")
if response.status_code == 403:
raise HTTPException(403, "Connect your own active LiteLLM proxy-admin account with the same email as Slack")
if response.status_code != 200:
raise HTTPException(503, "LiteAdmin could not verify this connection")
return response
async def details(self, token: str) -> LinkDetails:
response: Final = await self.worker_request(token)
try:
return LinkDetails.model_validate_json(response.content)
except ValidationError:
raise HTTPException(503, "LiteAdmin could not verify this connection") from None
async def admin(self, user_id: str, details: LinkDetails) -> LiteLLM_UserTable:
user: Final = await self.load_user(user_id)
if (
user is None
or user.user_role != LitellmUserRoles.PROXY_ADMIN.value
or not user.user_email
or user.user_email.strip().casefold() != details.email.strip().casefold()
):
raise HTTPException(403, "Connect your own active LiteLLM proxy-admin account with the same email as Slack")
return user
def _page(title: str, body: str) -> HTMLResponse:
return HTMLResponse(
f'<!doctype html><html lang="en"><meta charset="utf-8">'
f'<meta name="viewport" content="width=device-width,initial-scale=1"><title>{html.escape(title)}</title>'
"<style>body{font:17px system-ui;color:#18252f;max-width:560px;margin:10vh auto;padding:24px}"
"p{line-height:1.6}button{font:inherit;border:0;border-radius:8px;padding:14px 20px;background:#5b3fd1;"
"color:white;cursor:pointer}small{color:#556}</style>"
f"<main><h1>{html.escape(title)}</h1>{body}</main></html>",
headers=_HEADERS,
)
def _cookie_name(token: str) -> str:
return _COOKIE + hashlib.sha256(token.encode()).hexdigest()[:16]
async def _session_user(request: Request) -> str | None:
from litellm.proxy._experimental.mcp_server.byok_oauth_endpoints import (
get_authenticated_browser_user_id,
)
return await get_authenticated_browser_user_id(request)
async def _load_user(user_id: str) -> LiteLLM_UserTable | None:
from litellm.proxy.auth.auth_checks import get_user_object
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
if prisma_client is None:
raise HTTPException(503, "LiteAdmin requires a database")
try:
return await get_user_object(
user_id=user_id,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
user_id_upsert=False,
check_db_only=True,
)
except UserNotFoundError:
return None
except Exception:
raise HTTPException(503, "LiteAdmin could not verify your current permissions") from None
def mint_admin_session(user: LiteLLM_UserTable) -> AdminSession:
from litellm.proxy.auth.auth_checks import LITELLM_SESSION_TOKEN_PREFIX
from litellm.proxy.common_utils.encrypt_decrypt_utils import encrypt_bearer_token
expires: Final = datetime.now(timezone.utc) + timedelta(hours=24)
auth: Final = UserAPIKeyAuth(
token="liteadmin-" + secrets.token_urlsafe(24),
key_name="LiteAdmin Slack",
key_alias="LiteAdmin Slack",
user_id=user.user_id,
user_role=LitellmUserRoles.PROXY_ADMIN,
models=TypeAdapter(list[str]).validate_python(user.model_dump().get("models", [])),
expires=expires,
is_session_token=True,
)
return AdminSession(
user_id=user.user_id,
credential=SecretStr(
encrypt_bearer_token(auth.model_dump_json(exclude_none=True), LITELLM_SESSION_TOKEN_PREFIX)
),
expires_at=expires.timestamp(),
)
def validate_native_configuration(
worker_url: str, service_token: str, enterprise: bool, database_available: bool
) -> None:
if not worker_url:
raise HTTPException(404, "LiteAdmin Slack is not enabled")
if not enterprise:
raise HTTPException(403, "LiteAdmin Slack requires LiteLLM Enterprise")
if not database_available:
raise HTTPException(503, "LiteAdmin requires a database")
try:
parsed: Final = urlsplit(worker_url)
port: Final = parsed.port
except ValueError:
raise HTTPException(503, "LiteAdmin worker configuration is invalid") from None
if (
parsed.scheme not in {"http", "https"}
or not parsed.hostname
or port == 0
or parsed.username
or parsed.password
or parsed.path
or parsed.query
or parsed.fragment
or len(service_token) < 32
or any(character.isspace() for character in service_token)
):
raise HTTPException(503, "LiteAdmin worker configuration is invalid")
async def native_admin_context() -> NativeAdminContext:
from litellm.proxy.proxy_server import premium_user, prisma_client
worker_url: Final = os.getenv("LITELLM_ADMIN_AGENT_URL", "").rstrip("/")
service_token: Final = os.getenv("ADMIN_AGENT_SERVICE_TOKEN", "")
validate_native_configuration(worker_url, service_token, premium_user is True, prisma_client is not None)
client: Final = get_async_httpx_client(
llm_provider="liteadmin_native", params={"timeout": 15.0, "follow_redirects": False}
).client
return NativeAdminContext(
worker_url, SecretStr(service_token), client, _session_user, _load_user, mint_admin_session
)
@router.get(_PREFIX + "{token}", include_in_schema=False, response_class=HTMLResponse)
async def connect_page(
request: Request,
token: str,
context: Annotated[NativeAdminContext, Depends(native_admin_context)],
) -> Response:
details: Final = await context.details(token)
base_url: Final = get_request_base_url(request)
parsed_base: Final = urlsplit(base_url)
if parsed_base.scheme != "https":
raise HTTPException(400, "LiteAdmin account connections require HTTPS")
user_id: Final = await context.session_user(request)
if user_id is None:
return RedirectResponse(
base_url + "/sso/key/generate?" + urlencode({"return_to": parsed_base.path + _PREFIX + token}),
status_code=303,
headers=_HEADERS,
)
await context.admin(user_id, details)
csrf: Final = secrets.token_urlsafe(32)
page: Final = _page(
"Connect LiteAdmin to Slack",
f"<p>Connect <strong>{html.escape(details.email)}</strong> to LiteAdmin in your Slack workspace?</p>"
"<p>Model requests and administrative actions will use your own LiteLLM account and current permissions</p>"
f'<form method="post"><input type="hidden" name="csrf" value="{csrf}">'
'<button type="submit">Connect account</button></form>'
"<p><small>This connection lasts 24 hours. Send disconnect in Slack to remove the saved session</small></p>",
)
page.set_cookie(_cookie_name(token), csrf, max_age=600, secure=True, httponly=True, samesite="strict", path="/")
return page
@router.post(_PREFIX + "{token}", include_in_schema=False, response_class=HTMLResponse)
async def connect_account(
request: Request,
token: str,
context: Annotated[NativeAdminContext, Depends(native_admin_context)],
) -> Response:
base_url: Final = get_request_base_url(request)
parsed_base: Final = urlsplit(base_url)
origin: Final = f"{parsed_base.scheme}://{parsed_base.netloc}"
if parsed_base.scheme != "https" or request.headers.get("Origin") != origin:
raise HTTPException(403, "Reopen your private Slack connection link")
if request.headers.get("Content-Type", "").split(";", 1)[0] != "application/x-www-form-urlencoded":
raise HTTPException(400, "Expected a connection form")
form: Final = await request.form(max_fields=1, max_files=0, max_part_size=1024)
supplied: Final = form.get("csrf")
expected: Final = request.cookies.get(_cookie_name(token), "")
if (
not isinstance(supplied, str)
or len(expected) != 43
or len(supplied) != 43
or not hmac.compare_digest(supplied.encode(), expected.encode())
):
raise HTTPException(403, "Reopen your private Slack connection link")
user_id: Final = await context.session_user(request)
if user_id is None:
raise HTTPException(401, "Your login expired. Reopen your private Slack connection link")
details: Final = await context.details(token)
user: Final = await context.admin(user_id, details)
await context.worker_request(token, context.mint_session(user))
page: Final = _page(
"Account connected", "<p>Return to Slack and ask LiteAdmin to list your teams or check a budget</p>"
)
page.delete_cookie(_cookie_name(token), path="/", secure=True, httponly=True, samesite="strict")
return page

View file

@ -1,7 +1,7 @@
"""Path allowlist for the gateway component.
The gateway exposes the LLM data-plane surface: chat/completions, embeddings,
audio, batches, files, fine-tuning, rerank, ocr, rag, video, search, image,
audio, batches, files, fine-tuning, rerank, decisions, ocr, rag, video, search, image,
responses, vector stores, passthrough providers, realtime websockets, MCP
tool-call endpoints, and operational endpoints (/health, /metrics, and the
/debug/memory/summary read of the serving worker's RSS).
@ -60,6 +60,8 @@ GATEWAY_PATH_PREFIXES: tuple[str, ...] = (
"/v1/rerank",
"/v2/rerank",
"/rerank",
"/v1/decisions",
"/decisions",
"/v1/ocr",
"/ocr",
"/v1/rag/",

View file

@ -57,6 +57,19 @@ spec:
imagePullPolicy: {{ .Values.image.pullPolicy }}
env:
{{- include "litellm.proxyEnv" . | nindent 12 }}
{{- if .Values.liteadmin.enabled }}
- name: LITELLM_ADMIN_AGENT_URL
value: {{ printf "http://%s-liteadmin:10000" (include "litellm.fullname" . | trunc 53 | trimSuffix "-") | quote }}
- name: ADMIN_AGENT_SERVICE_TOKEN
valueFrom:
secretKeyRef:
name: {{ required "liteadmin.existingSecret is required" .Values.liteadmin.existingSecret }}
key: ADMIN_AGENT_SERVICE_TOKEN
{{- if not (hasKey (default dict .Values.envVars) "PROXY_BASE_URL") }}
- name: PROXY_BASE_URL
value: {{ required "liteadmin.gatewayUrl is required" .Values.liteadmin.gatewayUrl | quote }}
{{- end }}
{{- end }}
{{- include "litellm.proxyMetricsEnv" . | nindent 12 }}
{{- if .Values.collector.enabled }}
{{- include "litellm.collectorEnv" . | nindent 12 }}

View file

@ -0,0 +1,112 @@
{{- if .Values.liteadmin.enabled }}
{{- $name := printf "%s-liteadmin" (include "litellm.fullname" . | trunc 53 | trimSuffix "-") }}
{{- $secret := required "liteadmin.existingSecret is required" .Values.liteadmin.existingSecret }}
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ $name }}
spec:
replicas: 1
strategy:
type: Recreate
selector:
matchLabels:
app.kubernetes.io/name: {{ $name }}
app.kubernetes.io/instance: {{ .Release.Name }}
template:
metadata:
labels:
app.kubernetes.io/name: {{ $name }}
app.kubernetes.io/instance: {{ .Release.Name }}
spec:
automountServiceAccountToken: false
terminationGracePeriodSeconds: 75
{{- with .Values.imagePullSecrets }}
imagePullSecrets:
{{- toYaml . | nindent 8 }}
{{- end }}
securityContext:
runAsUser: 10001
runAsGroup: 10001
fsGroup: 10001
runAsNonRoot: true
containers:
- name: liteadmin
image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default .Chart.AppVersion }}"
imagePullPolicy: {{ .Values.image.pullPolicy }}
args: ["--admin-agent"]
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: [ALL]
envFrom:
- secretRef:
name: {{ $secret }}
env:
- name: CONNECTION_AUTH_MODE
value: native
- name: LITELLM_BASE_URL
value: {{ required "liteadmin.gatewayUrl is required" .Values.liteadmin.gatewayUrl | quote }}
- name: LITELLM_MODEL
value: {{ required "liteadmin.model is required" .Values.liteadmin.model | quote }}
- name: STATE_DB
value: /var/data/events.sqlite3
- name: ADMIN_READ_ONLY
value: {{ .Values.liteadmin.readOnly | quote }}
- name: OPENAI_AGENTS_DISABLE_TRACING
value: "1"
ports:
- name: health
containerPort: 10000
readinessProbe:
httpGet:
path: /readyz
port: health
periodSeconds: 15
livenessProbe:
httpGet:
path: /healthz
port: health
periodSeconds: 30
resources:
{{- toYaml .Values.liteadmin.resources | nindent 12 }}
volumeMounts:
- name: state
mountPath: /var/data
- name: tmp
mountPath: /tmp
volumes:
- name: state
persistentVolumeClaim:
claimName: {{ $name }}
- name: tmp
emptyDir:
sizeLimit: 64Mi
---
apiVersion: v1
kind: Service
metadata:
name: {{ $name }}
spec:
type: ClusterIP
selector:
app.kubernetes.io/name: {{ $name }}
app.kubernetes.io/instance: {{ .Release.Name }}
ports:
- port: 10000
targetPort: health
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: {{ $name }}
spec:
accessModes: [ReadWriteOnce]
{{- with .Values.liteadmin.storageClassName }}
storageClassName: {{ . | quote }}
{{- end }}
resources:
requests:
storage: {{ .Values.liteadmin.storageSize }}
{{- end }}

View file

@ -3,6 +3,20 @@
# Declare variables to be passed into your templates.
replicaCount: 1
liteadmin:
enabled: false
existingSecret: ""
gatewayUrl: ""
model: ""
readOnly: false
storageSize: 1Gi
storageClassName: ""
resources:
requests:
cpu: 100m
memory: 256Mi
limits:
memory: 1Gi
# numWorkers: 2
image:

View file

@ -471,6 +471,24 @@ Directory of the collector's unix socket, shared by the gateway and
collector containers through an emptyDir. Empty when the sidecar is off
or gateway.collector.address is a tcp://127.0.0.1:<port> address.
*/}}
{{- define "litellm.lensWorker.image" -}}
{{- if .Values.lensWorker.image.digest -}}
{{- if not (regexMatch "^sha256:[0-9a-f]{64}$" .Values.lensWorker.image.digest) -}}
{{- fail "lensWorker.image.digest must be sha256 followed by 64 lowercase hex characters" -}}
{{- end -}}
{{- printf "%s@%s" .Values.lensWorker.image.repository .Values.lensWorker.image.digest -}}
{{- else -}}
{{- $backendTag := .Values.backend.image.tag | default .Chart.AppVersion -}}
{{- $releaseTag := ternary (printf "v%s" $backendTag) $backendTag (regexMatch "^[0-9]" $backendTag) -}}
{{- $tag := .Values.lensWorker.image.tag | default $releaseTag -}}
{{- $repository := .Values.lensWorker.image.repository -}}
{{- if and (hasPrefix "sha-" $tag) (eq $repository "ghcr.io/berriai/litellm-lens-worker") -}}
{{- $repository = "ghcr.io/berriai/litellm-lens-worker-dev" -}}
{{- end -}}
{{- printf "%s:%s" $repository $tag -}}
{{- end -}}
{{- end -}}
{{- define "litellm.gateway.collectorSocketDir" -}}
{{- if and .Values.gateway.collector.enabled (hasPrefix "unix://" .Values.gateway.collector.address) -}}
{{- dir (trimPrefix "unix://" .Values.gateway.collector.address) -}}

View file

@ -57,6 +57,8 @@ spec:
containerPort: 4001
protocol: TCP
env:
- name: LENS_WORKER_IMAGE
value: {{ include "litellm.lensWorker.image" . | quote }}
{{- include "litellm.serverEnv" (dict "root" $ "component" .Values.backend) | nindent 12 }}
{{- if .Values.gateway.config.create }}
- name: CONFIG_FILE_PATH

View file

@ -0,0 +1,72 @@
{{- if .Values.lensWorker.enabled }}
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ include "litellm.fullname" . }}-lens-worker
labels:
{{- include "litellm.commonLabels" . | nindent 4 }}
app.kubernetes.io/component: lens-worker
spec:
replicas: {{ .Values.lensWorker.replicaCount }}
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: lens-worker
template:
metadata:
labels:
{{- include "litellm.commonLabels" . | nindent 8 }}
app.kubernetes.io/component: lens-worker
spec:
automountServiceAccountToken: false
{{- with .Values.imagePullSecrets }}
imagePullSecrets:
{{- toYaml . | nindent 8 }}
{{- end }}
securityContext:
runAsNonRoot: true
runAsUser: 65532
runAsGroup: 65532
fsGroup: 65532
seccompProfile:
type: RuntimeDefault
containers:
- name: lens-worker
image: {{ include "litellm.lensWorker.image" . | quote }}
imagePullPolicy: {{ .Values.lensWorker.image.pullPolicy }}
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: [ALL]
env:
- name: LITELLM_URL
value: {{ .Values.lensWorker.url | default (printf "http://%s:%v" (include "litellm.backend.fullname" .) .Values.backend.service.port) | quote }}
- name: LENS_WORKER_TOKEN
valueFrom:
secretKeyRef:
name: {{ required "lensWorker.tokenSecret.name must reference a Lens worker token" .Values.lensWorker.tokenSecret.name | quote }}
key: {{ .Values.lensWorker.tokenSecret.key | quote }}
resources:
{{- toYaml .Values.lensWorker.resources | nindent 12 }}
volumeMounts:
- name: tmp
mountPath: /tmp
volumes:
- name: tmp
emptyDir:
medium: Memory
sizeLimit: {{ .Values.lensWorker.tmpSizeLimit }}
{{- with .Values.lensWorker.nodeSelector }}
nodeSelector:
{{- toYaml . | nindent 8 }}
{{- end }}
{{- with .Values.lensWorker.tolerations }}
tolerations:
{{- toYaml . | nindent 8 }}
{{- end }}
{{- with .Values.lensWorker.affinity }}
affinity:
{{- toYaml . | nindent 8 }}
{{- end }}
{{- end }}

View file

@ -0,0 +1,174 @@
suite: Lens worker release and credentials
templates:
- lens/deployment.yaml
- backend/deployment.yaml
- gateway/configmap.yaml
values:
- ./values/required.yaml
tests:
- it: installs the development package for a source commit
template: lens/deployment.yaml
set:
backend.image.tag: sha-0123456789abcdef
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker-dev:sha-0123456789abcdef
- it: advertises the development package for standalone source workers
template: backend/deployment.yaml
set:
backend.image.tag: sha-0123456789abcdef
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: LENS_WORKER_IMAGE
value: ghcr.io/berriai/litellm-lens-worker-dev:sha-0123456789abcdef
- it: preserves an explicit private source image repository
template: lens/deployment.yaml
set:
backend.image.tag: sha-0123456789abcdef
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
lensWorker.image.repository: registry.example/lens-worker
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: registry.example/lens-worker:sha-0123456789abcdef
- it: pins the worker to its approved digest even when its tag changes
template: lens/deployment.yaml
set:
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
lensWorker.image.tag: replaced-release
lensWorker.image.digest: sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
- it: advertises the approved digest to standalone installers
template: backend/deployment.yaml
set:
lensWorker.image.tag: replaced-release
lensWorker.image.digest: sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: LENS_WORKER_IMAGE
value: ghcr.io/berriai/litellm-lens-worker@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
- it: refuses a malformed digest instead of falling back to the tag
template: backend/deployment.yaml
set:
lensWorker.image.digest: sha256:invalid
asserts:
- failedTemplate:
errorMessage: lensWorker.image.digest must be sha256 followed by 64 lowercase hex characters
- it: keeps the worker opt in
template: lens/deployment.yaml
asserts:
- hasDocuments:
count: 0
- it: requires a limited worker credential when enabled
template: lens/deployment.yaml
set:
lensWorker.enabled: true
asserts:
- failedTemplate:
errorMessage: lensWorker.tokenSecret.name must reference a Lens worker token
- it: uses the chart release and a secret without granting Kubernetes access
template: lens/deployment.yaml
chart:
appVersion: v1.2.3
set:
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker:v1.2.3
- equal:
path: spec.template.spec.containers[0].env[1].valueFrom.secretKeyRef
value:
name: lens-credential
key: token
- equal:
path: spec.template.spec.automountServiceAccountToken
value: false
- equal:
path: spec.template.spec.containers[0].securityContext.readOnlyRootFilesystem
value: true
- equal:
path: spec.template.spec.volumes[0].emptyDir
value:
medium: Memory
sizeLimit: 1Gi
- it: advertises the same private dev image to standalone installers
template: backend/deployment.yaml
set:
lensWorker.image.repository: registry.example/lens-worker
lensWorker.image.tag: branch-main-1234567
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: LENS_WORKER_IMAGE
value: registry.example/lens-worker:branch-main-1234567
- it: supports an external gateway and a registry override
template: lens/deployment.yaml
set:
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
lensWorker.url: https://gateway.example/proxy
lensWorker.image.repository: registry.example/lens-worker
lensWorker.image.tag: branch-main-1234567
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: registry.example/lens-worker:branch-main-1234567
- equal:
path: spec.template.spec.containers[0].env[0].value
value: https://gateway.example/proxy
- it: prefixes a numeric chart release with v
template: lens/deployment.yaml
chart:
appVersion: 1.2.3-rc.4
set:
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker:v1.2.3-rc.4
- it: follows a backend image override when no worker tag is set
template: lens/deployment.yaml
set:
backend.image.tag: branch-main-1234567
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker:branch-main-1234567
- it: recommends the overridden backend release for standalone installers
template: backend/deployment.yaml
set:
backend.image.tag: v1.2.3-dev.4
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: LENS_WORKER_IMAGE
value: ghcr.io/berriai/litellm-lens-worker:v1.2.3-dev.4
- it: normalizes a numeric backend tag to the published worker tag
template: lens/deployment.yaml
set:
backend.image.tag: 1.2.3-dev.4
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker:v1.2.3-dev.4

View file

@ -629,3 +629,26 @@ ui:
affinity: {}
# Same shape as gateway.topologySpreadConstraints.
topologySpreadConstraints: []
lensWorker:
enabled: false
replicaCount: 1
image:
repository: ghcr.io/berriai/litellm-lens-worker
tag: ""
digest: ""
pullPolicy: IfNotPresent
tokenSecret:
name: ""
key: token
url: ""
tmpSizeLimit: 1Gi
resources:
requests:
cpu: 100m
memory: 256Mi
limits:
memory: 2Gi
nodeSelector: {}
tolerations: []
affinity: {}

View file

@ -1,3 +1,4 @@
import functools
import glob
import os
import random
@ -10,7 +11,8 @@ import time
from collections.abc import Callable
from dataclasses import dataclass, replace
from pathlib import Path
from typing import TYPE_CHECKING, Final, Optional
from typing import TYPE_CHECKING, Final, Optional, Union
from urllib.parse import unquote, urlsplit
from litellm_proxy_extras import prisma_toolchain
from litellm_proxy_extras._logging import logger
@ -202,6 +204,66 @@ def _max_migration_timestamp(names) -> int:
return max(_migration_timestamp(n) for n in names)
_REDACTED: Final = "REDACTED"
_PASSWORD_QUERY_KEYS: Final = frozenset(("password", "sslpassword"))
@functools.cache
def _secret_shape_redactor() -> Callable[[str], str]:
try:
from litellm._logging import redact_secrets
except ImportError:
return lambda text: text
return redact_secrets
def _url_passwords(url: str) -> frozenset[str]:
try:
parts: Final = urlsplit(url)
except ValueError:
return frozenset()
query_pairs: Final = tuple(pair.partition("=") for pair in parts.query.split("&"))
raw_query_passwords: Final = tuple(
value for key, separator, value in query_pairs if separator and key.lower() in _PASSWORD_QUERY_KEYS
)
raw_passwords: Final = ((parts.password,) if parts.password else ()) + raw_query_passwords
return frozenset(password for password in raw_passwords + tuple(map(unquote, raw_passwords)) if password)
def _configured_database_passwords() -> frozenset[str]:
database_url: Final = os.getenv("DATABASE_URL")
direct_url: Final = os.getenv("DIRECT_URL")
database_passwords: Final = _url_passwords(database_url) if database_url else frozenset()
direct_passwords: Final = _url_passwords(direct_url) if direct_url else frozenset()
return database_passwords | direct_passwords
def _redact_credentials(text: str) -> str:
"""Mask configured database passwords before passing the text to LiteLLM redaction."""
passwords: Final = sorted(_configured_database_passwords(), key=len, reverse=True)
alternation: Final = "|".join(re.escape(password) for password in passwords)
password_pattern: Final = (
re.compile(rf"(?P<lead>:|password=)(?:{alternation})(?=@|&|$|[\s'\"\]),])", re.IGNORECASE)
if passwords
else None
)
result: Final = password_pattern.sub(rf"\g<lead>{_REDACTED}", text) if password_pattern is not None else text
return _secret_shape_redactor()(result)
def _redacted_command(command: object) -> Union[str, tuple[str, ...], list[str]]:
if isinstance(command, tuple):
return tuple(_redact_credentials(str(argument)) for argument in command)
if isinstance(command, list):
return [_redact_credentials(str(argument)) for argument in command]
return _redact_credentials(str(command))
def _redact_command_error(error: subprocess.CalledProcessError) -> str:
redacted_command: Final = _redacted_command(error.cmd)
return str(subprocess.CalledProcessError(error.returncode, redacted_command))
def _get_prisma_command() -> str:
"""Get the Prisma command to use, bypassing Python wrapper in offline mode."""
if str_to_bool(os.getenv("PRISMA_OFFLINE_MODE")):
@ -315,7 +377,8 @@ class ProxyExtrasDBManager:
return False
except subprocess.CalledProcessError as e:
logger.warning(
f"Error creating baseline migration: {e}, {e.stderr}, {e.stdout}"
f"Error creating baseline migration: {_redact_command_error(e)}, "
f"{_redact_credentials(str(e.stderr))}, {_redact_credentials(str(e.stdout))}"
)
raise e
@ -1572,6 +1635,11 @@ class ProxyExtrasDBManager:
f"Error: {stderr}"
)
raise
else:
logger.error(
"prisma migrate deploy failed with an error the resolver does not handle: "
f"{_redact_credentials(stderr)}"
)
else:
if ProxyExtrasDBManager.spend_logs_is_partitioned():
raise RuntimeError(PARTITIONED_SPEND_LOGS_PUSH_ERROR)
@ -1586,7 +1654,7 @@ class ProxyExtrasDBManager:
)
return True
except subprocess.TimeoutExpired:
logger.warning(
logger.error(
"Attempt %s timed out. Raise %s if this database needs longer to apply its schema.",
attempt + 1,
PRISMA_MIGRATE_DEPLOY_TIMEOUT_ENV_VAR if use_migrate else PRISMA_COMMAND_TIMEOUT_ENV_VAR,
@ -1599,7 +1667,12 @@ class ProxyExtrasDBManager:
if attempts_left > 0
else ""
)
logger.info(f"The process failed to execute. Details: {e}.{retry_msg}")
stderr_detail: Final = (
f" stderr: {_redact_credentials(str(e.stderr))}" if e.stderr else ""
)
logger.error(
f"The process failed to execute. Details: {_redact_command_error(e)}.{stderr_detail}{retry_msg}"
)
time.sleep(random.randrange(5, 15))
finally:
os.chdir(original_dir)

View file

@ -4085,26 +4085,6 @@ dependencies = [
"strum",
]
[[package]]
name = "litellm-migrate"
version = "0.1.0"
dependencies = [
"litellm-migrate-macros",
"rstest",
]
[[package]]
name = "litellm-migrate-macros"
version = "0.1.0"
dependencies = [
"proc-macro2",
"quote",
"rstest",
"syn 2.0.119",
"tempfile",
"thiserror 2.0.19",
]
[[package]]
name = "litellm-model-catalog"
version = "0.1.0"
@ -4156,6 +4136,7 @@ dependencies = [
"litellm-storage-clickhouse",
"litellm-token-counter",
"litellm-traces",
"litellm-traces-cache",
"litellm-traces-clickhouse",
"litellm-tracing",
"prost",
@ -4169,6 +4150,7 @@ dependencies = [
"serde_json",
"serde_with",
"sha2 0.10.9",
"sqlx",
"strum",
"thiserror 2.0.19",
"tokio",
@ -4367,6 +4349,9 @@ dependencies = [
"rstest",
"serde",
"serde_json",
"serde_with",
"sqlx",
"testcontainers-modules",
"thiserror 2.0.19",
"tokio",
"url",
@ -4454,6 +4439,7 @@ name = "litellm-traces"
version = "0.1.0"
dependencies = [
"askama",
"base64 0.22.1",
"criterion",
"indexmap 2.14.0",
"litellm-llms-types",
@ -4469,20 +4455,36 @@ dependencies = [
"time",
]
[[package]]
name = "litellm-traces-cache"
version = "0.1.0"
dependencies = [
"base64 0.22.1",
"litellm-traces",
"moka",
"rstest",
"serde",
"serde_json",
"sha2 0.10.9",
"thiserror 2.0.19",
"time",
"tokio",
"tracing",
]
[[package]]
name = "litellm-traces-clickhouse"
version = "0.1.0"
dependencies = [
"askama",
"base64 0.22.1",
"flate2",
"futures-util",
"hmac 0.12.1",
"jsonschema",
"litellm-http",
"litellm-migrate",
"litellm-storage-clickhouse",
"litellm-traces",
"litellm-traces-cache",
"macro_rules_attribute",
"moka",
"rstest",
@ -4490,12 +4492,12 @@ dependencies = [
"serde",
"serde_json",
"sha2 0.10.9",
"sqlx",
"strum",
"testcontainers-modules",
"thiserror 2.0.19",
"time",
"tokio",
"tracing",
"url",
"wiremock",
]

View file

@ -13,10 +13,9 @@ litellm-config = { path = "crates/config" }
litellm-router = { path = "crates/router" }
litellm-tracing = { path = "crates/tracing" }
litellm-traces = { path = "crates/traces" }
litellm-traces-cache = { path = "crates/traces-cache" }
litellm-traces-clickhouse = { path = "crates/traces-clickhouse" }
litellm-storage-clickhouse = { path = "crates/storage-clickhouse" }
litellm-migrate = { path = "crates/migrate" }
litellm-migrate-macros = { path = "crates/migrate-macros" }
litellm-core = { path = "crates/core" }
litellm-gateway-mcp = { path = "crates/gateway-mcp" }
litellm-gateway = { path = "crates/gateway" }

View file

@ -1,19 +0,0 @@
[package]
name = "litellm-migrate-macros"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[lib]
proc-macro = true
[dependencies]
proc-macro2.workspace = true
quote.workspace = true
syn = { workspace = true, features = ["parsing", "printing", "proc-macro"] }
thiserror.workspace = true
[dev-dependencies]
rstest.workspace = true
tempfile.workspace = true

View file

@ -1,21 +0,0 @@
use std::io;
#[derive(Debug, thiserror::Error)]
pub enum Error {
#[error("could not read migrations directory `{path}`")]
ReadDirectory {
path: String,
#[source]
source: io::Error,
},
#[error(
"migration name `{name}` must be `<digits>_<description>.sql` with a `[a-z0-9_]` description"
)]
InvalidName { name: String },
#[error("migration version `{version}` is declared more than once")]
DuplicateVersion { version: u64 },
#[error("migrations directory `{path}` contains no migrations")]
Empty { path: String },
#[error("migration path `{path}` is not valid UTF-8")]
NonUtf8Path { path: String },
}

View file

@ -1,199 +0,0 @@
mod error;
use std::path::{Path, PathBuf};
use error::Error;
use proc_macro::TokenStream;
use quote::quote;
use syn::LitStr;
struct Entry {
version: u64,
description: String,
path: PathBuf,
}
fn resolve(dir: &Path) -> Result<Vec<Entry>, Error> {
let mut entries = Vec::new();
let files = std::fs::read_dir(dir).map_err(|source| Error::ReadDirectory {
path: dir.display().to_string(),
source,
})?;
for file in files {
let file = file.map_err(|source| Error::ReadDirectory {
path: dir.display().to_string(),
source,
})?;
let path = file.path();
let name = path
.file_name()
.and_then(|name| name.to_str())
.ok_or_else(|| Error::NonUtf8Path {
path: path.display().to_string(),
})?
.to_owned();
let invalid = || Error::InvalidName { name: name.clone() };
let stem = name
.strip_suffix(".sql")
.filter(|_| file.file_type().is_ok_and(|kind| kind.is_file()))
.and_then(|stem| stem.split_once('_'))
.filter(|(version, description)| {
!version.is_empty()
&& version.bytes().all(|b| b.is_ascii_digit())
&& !description.is_empty()
&& description
.bytes()
.all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'_')
})
.ok_or_else(invalid)?;
let version = stem.0.parse::<u64>().map_err(|_| invalid())?;
entries.push(Entry {
version,
description: stem.1.to_owned(),
path,
});
}
if entries.is_empty() {
return Err(Error::Empty {
path: dir.display().to_string(),
});
}
entries.sort_by_key(|entry| entry.version);
for pair in entries.windows(2) {
if pair[0].version == pair[1].version {
return Err(Error::DuplicateVersion {
version: pair[0].version,
});
}
}
Ok(entries)
}
fn resolve_input(lit: &LitStr) -> Result<Vec<Entry>, Error> {
let root = std::env::var("CARGO_MANIFEST_DIR")
.map(PathBuf::from)
.unwrap_or_default();
let dir = root.join(lit.value());
let dir = dir.canonicalize().map_err(|source| Error::ReadDirectory {
path: dir.display().to_string(),
source,
})?;
if dir.to_str().is_none() {
return Err(Error::NonUtf8Path {
path: dir.display().to_string(),
});
}
resolve(&dir)
}
#[proc_macro]
pub fn migrate(input: TokenStream) -> TokenStream {
let lit = syn::parse_macro_input!(input as LitStr);
match resolve_input(&lit) {
Ok(entries) => {
let migrations = entries.iter().map(|entry| {
let version = entry.version;
let description = &entry.description;
let path = entry
.path
.to_str()
.expect("canonical migration path is UTF-8");
quote! {
::litellm_migrate::Migration {
version: #version,
description: #description,
sql: ::core::include_str!(#path),
}
}
});
quote! { &[#(#migrations),*] }.into()
}
Err(err) => syn::Error::new(lit.span(), err).to_compile_error().into(),
}
}
#[cfg(test)]
mod tests {
use std::fs;
use rstest::rstest;
use tempfile::TempDir;
use super::{Error, resolve};
fn migrations_dir(files: &[&str]) -> TempDir {
let dir = TempDir::new().expect("tempdir");
for file in files {
fs::write(dir.path().join(file), "SELECT 1").expect("write fixture");
}
dir
}
#[rstest]
fn orders_versions_numerically() {
let dir = migrations_dir(&["10_tenth.sql", "2_second.sql", "1_first.sql"]);
let entries = resolve(dir.path()).expect("resolves");
let versions: Vec<u64> = entries.iter().map(|entry| entry.version).collect();
let descriptions: Vec<&str> = entries
.iter()
.map(|entry| entry.description.as_str())
.collect();
assert_eq!(versions, [1, 2, 10]);
assert_eq!(descriptions, ["first", "second", "tenth"]);
}
#[rstest]
#[case::dash_in_version(&["0001-dash.sql"])]
#[case::not_sql(&["notes.txt"])]
#[case::empty_description(&["0001_.sql"])]
#[case::non_digit_version(&["x_name.sql"])]
#[case::uppercase_description(&["0001_Upper.sql"])]
#[case::no_underscore(&["0001.sql"])]
#[case::plus_sign_version(&["+10_add.sql"])]
fn rejects_invalid_names(#[case] files: &[&str]) {
let dir = migrations_dir(files);
assert!(matches!(
resolve(dir.path()),
Err(Error::InvalidName { .. })
));
}
#[rstest]
fn rejects_subdirectories() {
let dir = migrations_dir(&["0001_a.sql"]);
fs::create_dir(dir.path().join("0002_b.sql")).expect("subdir");
assert!(matches!(
resolve(dir.path()),
Err(Error::InvalidName { .. })
));
}
#[cfg(unix)]
#[rstest]
fn rejects_symlinks() {
let dir = migrations_dir(&["0001_a.sql"]);
let target = TempDir::new().expect("tempdir");
let target_file = target.path().join("real.sql");
fs::write(&target_file, "SELECT 2").expect("write fixture");
std::os::unix::fs::symlink(&target_file, dir.path().join("0002_b.sql")).expect("symlink");
assert!(matches!(
resolve(dir.path()),
Err(Error::InvalidName { .. })
));
}
#[rstest]
fn rejects_duplicate_versions() {
let dir = migrations_dir(&["0001_a.sql", "1_b.sql"]);
assert!(matches!(
resolve(dir.path()),
Err(Error::DuplicateVersion { version: 1 })
));
}
#[rstest]
fn rejects_empty_directory() {
let dir = migrations_dir(&[]);
assert!(matches!(resolve(dir.path()), Err(Error::Empty { .. })));
}
}

View file

@ -1,12 +0,0 @@
[package]
name = "litellm-migrate"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
litellm-migrate-macros.workspace = true
[dev-dependencies]
rstest.workspace = true

View file

@ -1,5 +0,0 @@
# Migrations
`litellm-migrate` exports the `Migration` struct and the `migrate!` macro that embeds a directory of `<digits>_<description>.sql` files at compile time, sorted by numeric version
The crate does not apply or track migrations; callers decide how and when the embedded SQL runs

View file

@ -1,8 +0,0 @@
pub use litellm_migrate_macros::migrate;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct Migration {
pub version: u64,
pub description: &'static str,
pub sql: &'static str,
}

View file

@ -1 +0,0 @@
SELECT 10;

View file

@ -1 +0,0 @@
SELECT 1;

View file

@ -1 +0,0 @@
SELECT 2;

View file

@ -1,21 +0,0 @@
use litellm_migrate::Migration;
use rstest::rstest;
const MIGRATIONS: &[Migration] = litellm_migrate::migrate!("tests/fixtures/migrations");
#[rstest]
#[case::first(0, 1, "first", include_str!("fixtures/migrations/1_first.sql"))]
#[case::second(1, 2, "second", include_str!("fixtures/migrations/2_second.sql"))]
#[case::tenth(2, 10, "tenth", include_str!("fixtures/migrations/10_tenth.sql"))]
fn embeds_every_file_sorted_by_numeric_version(
#[case] index: usize,
#[case] version: u64,
#[case] description: &str,
#[case] sql: &str,
) {
assert_eq!(MIGRATIONS.len(), 3);
let migration = &MIGRATIONS[index];
assert_eq!(migration.version, version);
assert_eq!(migration.description, description);
assert_eq!(migration.sql, sql);
}

View file

@ -22,6 +22,7 @@ tiktoken = ["litellm-token-counter/tiktoken"]
fancy-regex.workspace = true
litellm-tracing.workspace = true
litellm-traces.workspace = true
litellm-traces-cache.workspace = true
litellm-traces-clickhouse.workspace = true
litellm-storage-clickhouse.workspace = true
litellm-host.workspace = true
@ -73,6 +74,7 @@ serde_with.workspace = true
criterion.workspace = true
futures-util.workspace = true
rstest.workspace = true
sqlx = { workspace = true, features = ["migrate"] }
sha2.workspace = true
tokio-tungstenite.workspace = true
wiremock.workspace = true

View file

@ -1,8 +1,11 @@
use std::collections::BTreeMap;
use std::{collections::BTreeMap, sync::Arc};
use litellm_http::ClientVariant;
use litellm_traces::{QueryScope, ReadQuery, Tenant, query::named::ReadAccessParams};
use litellm_traces_clickhouse::{Config, Error, InsertTable, Parameter, QueryReaders};
use litellm_traces_cache::{ReadError, TraceReader};
use litellm_traces_clickhouse::{
ClickHouseTraces, Config, Error, InsertTable, Parameter, QueryReaders,
};
use prost::Message;
use pyo3::{
exceptions::{PyOverflowError, PyRuntimeError, PyValueError},
@ -10,6 +13,8 @@ use pyo3::{
types::PyBytes,
};
pyo3::import_exception!(litellm.rust_bridge.trace.errors, TraceChanged);
#[derive(Message)]
struct OtlpErrorStatus {
#[prost(int32, tag = "1")]
@ -39,17 +44,15 @@ fn map_error_ref(error: &Error) -> PyErr {
PyOverflowError::new_err(error.to_string())
}
Error::InvalidRow
| Error::InvalidLimit(_)
| Error::InvalidTable
| Error::InvalidCursor(_)
| Error::AmbiguousTrace
| Error::Decode(_)
| Error::InvalidSchema
| Error::InvalidQuery
| Error::InvalidParameters
| Error::InvalidScope => PyValueError::new_err(error.to_string()),
Error::Task
| Error::SchemaFailed(_)
| Error::SchemaTransport
| Error::Migration(_)
| Error::MissingSecret
| Error::Busy
| Error::ProvisionFailed(_)
@ -58,6 +61,7 @@ fn map_error_ref(error: &Error) -> PyErr {
Error::Cached(source) => map_error_ref(source),
Error::Storage(source) => match source {
StorageError::InvalidRow
| StorageError::InvalidLimit(_)
| StorageError::InvalidTable
| StorageError::InvalidSchema
| StorageError::EmptySql
@ -75,6 +79,18 @@ fn map_error_ref(error: &Error) -> PyErr {
}
}
fn map_read_error(error: ReadError<Error>) -> PyErr {
match error {
error @ (ReadError::InvalidParameters
| ReadError::InvalidCursor(_)
| ReadError::AmbiguousTrace) => PyValueError::new_err(error.to_string()),
error @ ReadError::TraceChanged => TraceChanged::new_err(error.to_string()),
error @ ReadError::TooLarge => PyOverflowError::new_err(error.to_string()),
error @ ReadError::Encode(_) => PyRuntimeError::new_err(error.to_string()),
ReadError::Store(error) => map_error_ref(&error),
}
}
fn map_sql_error(error: Error) -> PyErr {
match error {
Error::Storage(litellm_storage_clickhouse::Error::QueryFailed(400 | 404)) => {
@ -109,6 +125,7 @@ impl NativeTraceConfig {
pub struct NativeTraceStorage {
config: Config,
query_readers: QueryReaders,
reader: Arc<TraceReader>,
}
#[pymethods]
@ -120,6 +137,9 @@ impl NativeTraceStorage {
config.inner.storage().writer().clone(),
config.inner.storage().database().to_owned(),
),
reader: Arc::new(TraceReader::new(
litellm_storage_clickhouse::READ_LIMITS.response_bytes,
)),
config: config.inner.clone(),
})
}
@ -215,46 +235,56 @@ impl NativeTraceStorage {
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::list_traces(
&client,
&connection,
&scope,
start_ms,
end_ms,
cursor.as_deref(),
limit,
)
.await
let store = ClickHouseTraces::new(client, connection);
reader
.list_traces(&store, &scope, start_ms, end_ms, cursor.as_deref(), limit)
.await
},
map_error,
map_read_error,
)
}
#[pyo3(signature = (trace_id, scope, trace_ref, cursor=None, page_size=None))]
fn get_trace<'py>(
&self,
py: Python<'py>,
trace_id: String,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
trace_ref: String,
cursor: Option<String>,
page_size: Option<u32>,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::get_trace(
&client,
&connection,
&scope,
&trace_id,
&trace_ref,
)
.await
let store = ClickHouseTraces::new(client, connection);
if let Some(page_size) = page_size {
reader
.get_trace_page(
&store,
&scope,
&trace_id,
&trace_ref,
cursor.as_deref(),
page_size,
)
.await
} else if cursor.is_some() {
Err(ReadError::InvalidParameters)
} else {
reader
.get_trace(&store, &scope, &trace_id, &trace_ref)
.await
}
},
map_error,
map_read_error,
)
}
@ -268,20 +298,16 @@ impl NativeTraceStorage {
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::get_span(
&client,
&connection,
&scope,
&trace_id,
&span_id,
&trace_ref,
)
.await
let store = ClickHouseTraces::new(client, connection);
reader
.get_span(&store, &scope, &trace_id, &span_id, &trace_ref)
.await
},
map_error,
map_read_error,
)
}
@ -297,21 +323,23 @@ impl NativeTraceStorage {
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::get_span_error(
&client,
&connection,
&scope,
&trace_id,
&span_id,
&trace_ref,
cursor.as_deref(),
)
.await
let store = ClickHouseTraces::new(client, connection);
reader
.get_span_error(
&store,
&scope,
&trace_id,
&span_id,
&trace_ref,
cursor.as_deref(),
)
.await
},
map_error,
map_read_error,
)
}
@ -414,9 +442,23 @@ mod tests {
#[rstest]
#[case::row(Error::InvalidRow, "ValueError")]
#[case::insert_limit(Error::InvalidLimit("CLICKHOUSE_TRACE_MAX_INSERT_BYTES"), "ValueError")]
#[case::insert_timeout(
Error::Storage(litellm_storage_clickhouse::Error::InvalidLimit(
"CLICKHOUSE_INSERT_TIMEOUT_SECONDS"
)),
"ValueError"
)]
#[case::insert_budget(Error::InsertTooLarge, "OverflowError")]
#[case::scope(Error::InvalidScope, "ValueError")]
#[case::schema(Error::SchemaFailed(503), "RuntimeError")]
#[case::schema(
Error::Storage(litellm_storage_clickhouse::Error::SchemaFailed(503)),
"RuntimeError"
)]
#[case::migration(
Error::Migration(sqlx::migrate::MigrateError::VersionMismatch(1)),
"RuntimeError"
)]
#[case::reader(Error::MissingSecret, "RuntimeError")]
#[case::storage(
Error::Storage(litellm_storage_clickhouse::Error::InvalidUrl),
@ -463,9 +505,11 @@ mod tests {
#[rstest]
#[case::decode_budget(Error::Decode(litellm_traces::Error::TooLarge), "OverflowError")]
#[case::invalid_export(Error::Decode(litellm_traces::Error::InvalidPayload), "ValueError")]
#[case::cursor(Error::InvalidCursor("trace"), "ValueError")]
#[case::ambiguous(Error::AmbiguousTrace, "ValueError")]
fn trace_read_and_ingest_failures_preserve_public_exception_types(
#[case::invalid_decode_limit(
Error::Decode(litellm_traces::Error::InvalidLimit("OTLP_MAX_SPANS")),
"ValueError"
)]
fn trace_ingest_failures_preserve_public_exception_types(
#[case] error: Error,
#[case] exception_name: &str,
) {
@ -477,4 +521,43 @@ mod tests {
);
});
}
#[rstest]
#[case::invalid_parameters(ReadError::InvalidParameters, "ValueError")]
#[case::invalid_cursor(ReadError::InvalidCursor("trace"), "ValueError")]
#[case::ambiguous(ReadError::AmbiguousTrace, "ValueError")]
#[case::changed_snapshot(ReadError::TraceChanged, "TraceChanged")]
#[case::read_budget(ReadError::TooLarge, "OverflowError")]
#[case::encode(
ReadError::Encode(Arc::new(serde_json::Error::io(std::io::Error::other("invalid")))),
"RuntimeError"
)]
#[case::store(ReadError::Store(Arc::new(Error::InvalidScope)), "ValueError")]
fn trace_read_failures_preserve_public_exception_types(
#[case] error: ReadError<Error>,
#[case] exception_name: &str,
) {
Python::initialize();
Python::attach(|py| {
let repository = std::path::Path::new(env!("CARGO_MANIFEST_DIR"))
.ancestors()
.nth(3)
.unwrap()
.to_str()
.unwrap();
pyo3::types::PyModule::import(py, "sys")
.unwrap()
.getattr("path")
.unwrap()
.call_method1("insert", (0, repository))
.unwrap();
let message = error.to_string();
let exception = map_read_error(error);
assert_eq!(exception.get_type(py).name().unwrap(), exception_name);
assert_eq!(
exception.value(py).str().unwrap().to_str().unwrap(),
message
);
});
}
}

View file

@ -3,3 +3,7 @@
`litellm-storage-clickhouse` exports `Storage`, a writer and bounded reader derived from one ClickHouse URL and database. It also exports bounded HTTP read and insert execution
The crate has no trace tables, OTLP types, or named trace queries. `litellm-traces-clickhouse` supplies those rules and uses this storage for both trace rows and spend rows
It also applies embedded SQLx migrations through the `_sqlx_migrations` ledger
Migration files are append-only, and changed applied files are rejected by their checksums. Startup migrations must be replay-safe schema changes because the runner records success after execution without dirty states or locks. Backfills belong in coordinated jobs outside proxy startup. The replay policy lives in `ClickHouseMigrate::apply`, `dirty_version`, and `lock`; a Keeper-backed or deploy-time runner changes only those methods

View file

@ -11,11 +11,14 @@ flate2.workspace = true
litellm-http.workspace = true
serde.workspace = true
serde_json.workspace = true
serde_with.workspace = true
sqlx = { workspace = true, features = ["migrate"] }
thiserror.workspace = true
url.workspace = true
[dev-dependencies]
litellm-http = { workspace = true, features = ["test-support"] }
rstest.workspace = true
testcontainers-modules = { version = "0.15.0", features = ["clickhouse"] }
tokio.workspace = true
wiremock.workspace = true

View file

@ -1,7 +1,9 @@
#[derive(Debug, thiserror::Error)]
#[derive(Clone, Debug, thiserror::Error)]
pub enum Error {
#[error("invalid ClickHouse insert row")]
InvalidRow,
#[error("{0} must be a positive integer")]
InvalidLimit(&'static str),
#[error("invalid ClickHouse insert table")]
InvalidTable,
#[error("invalid ClickHouse HTTP URL")]

View file

@ -5,7 +5,19 @@ use litellm_http::Client;
use crate::{Connection, Error, valid_identifier};
const INSERT_TIMEOUT: Duration = Duration::from_secs(30);
fn insert_timeout() -> Result<Duration, Error> {
let name = "CLICKHOUSE_INSERT_TIMEOUT_SECONDS";
match std::env::var(name) {
Ok(value) => value
.parse::<u64>()
.ok()
.filter(|value| *value > 0)
.map(Duration::from_secs)
.ok_or(Error::InvalidLimit(name)),
Err(std::env::VarError::NotPresent) => Ok(Duration::from_secs(30)),
Err(_) => Err(Error::InvalidLimit(name)),
}
}
pub async fn insert_encoded_rows(
client: &Client,
@ -74,7 +86,7 @@ pub async fn insert_compressed_rows(
.append_pair("date_time_input_format", "best_effort");
let response = client
.post(url)
.timeout(INSERT_TIMEOUT)
.timeout(insert_timeout()?)
.header("Content-Encoding", "gzip")
.body(body)
.send()

View file

@ -1,9 +1,11 @@
mod error;
mod insert;
mod migrate;
mod read;
pub use error::Error;
pub use insert::{insert_compressed_rows, insert_encoded_rows};
pub use migrate::{ClickHouseMigrate, execute_statement, storage_error};
pub use read::{Parameter, Query, READ_LIMITS, ReadLimits, execute_read, fetch, fetch_json};
use url::Url;

View file

@ -0,0 +1,308 @@
use std::{
future::Future,
pin::Pin,
time::{Duration, Instant},
};
use litellm_http::Client;
use serde::Deserialize;
use serde_with::{DisplayFromStr, PickFirst, serde_as};
use sqlx::{
Error as SqlxError,
migrate::{AppliedMigration, Migrate, MigrateError, Migration},
};
use crate::{Connection, Error, READ_LIMITS, valid_identifier};
pub async fn execute_statement(
client: &Client,
connection: &Connection,
sql: &str,
timeout: Duration,
) -> Result<(), Error> {
let body = execute_sql(client, connection, sql, timeout).await?;
if !body.trim().is_empty() {
return Err(Error::InvalidResponse);
}
Ok(())
}
async fn execute_sql(
client: &Client,
connection: &Connection,
sql: &str,
timeout: Duration,
) -> Result<String, Error> {
let mut url = connection.url().clone();
let pairs: Vec<_> = url
.query_pairs()
.filter(|(key, _)| {
!matches!(
key.as_ref(),
"query" | "wait_end_of_query" | "send_progress_in_http_headers" | "async_insert"
)
})
.map(|(key, value)| (key.into_owned(), value.into_owned()))
.collect();
url.query_pairs_mut()
.clear()
.extend_pairs(pairs)
.append_pair("wait_end_of_query", "1")
.append_pair("send_progress_in_http_headers", "0")
.append_pair("async_insert", "0");
let mut response = client
.post(url)
.timeout(timeout)
.body(sql.to_owned())
.send()
.await
.map_err(|_| Error::Transport)?;
if !response.status().is_success() {
return Err(Error::SchemaFailed(response.status().as_u16()));
}
let mut body = Vec::new();
while let Some(chunk) = response.chunk().await.map_err(|_| Error::Transport)? {
if body.len() + chunk.len() > READ_LIMITS.response_bytes {
return Err(Error::ResponseTooLarge);
}
body.extend_from_slice(&chunk);
}
String::from_utf8(body).map_err(|_| Error::InvalidResponse)
}
/// The startup runner records success after execution, never marks migrations dirty, and skips locks
/// A Keeper-backed or deploy-time runner changes only `apply`, `dirty_version`, and `lock`
pub struct ClickHouseMigrate<'a, R> {
client: &'a Client,
connection: &'a Connection,
database: &'a str,
render: R,
timeout: Duration,
}
impl<'a, R> ClickHouseMigrate<'a, R>
where
R: Fn(&str) -> String + Send + Sync,
{
pub fn new(
client: &'a Client,
connection: &'a Connection,
database: &'a str,
render: R,
timeout: Duration,
) -> Result<Self, Error> {
if !valid_identifier(database) {
return Err(Error::InvalidSchema);
}
Ok(Self {
client,
connection,
database,
render,
timeout,
})
}
}
#[serde_as]
#[derive(Deserialize)]
struct Applied {
#[serde_as(as = "PickFirst<(_, DisplayFromStr)>")]
version: i64,
checksum: String,
}
fn migrate_error(error: Error) -> MigrateError {
MigrateError::Execute(SqlxError::AnyDriverError(Box::new(error)))
}
fn migrate_execution_error(error: Error, version: i64) -> MigrateError {
MigrateError::ExecuteMigration(SqlxError::AnyDriverError(Box::new(error)), version)
}
fn encode_hex(bytes: &[u8]) -> String {
bytes
.iter()
.map(|byte| format!("{byte:02x}"))
.collect::<Vec<_>>()
.join("")
}
fn decode_hex(value: &str) -> Result<Vec<u8>, Error> {
let (pairs, remainder) = value.as_bytes().as_chunks::<2>();
if !remainder.is_empty() {
return Err(Error::InvalidResponse);
}
pairs
.iter()
.map(|pair| {
let high = decode_hex_digit(pair[0]).ok_or(Error::InvalidResponse)?;
let low = decode_hex_digit(pair[1]).ok_or(Error::InvalidResponse)?;
Ok((high << 4) | low)
})
.collect()
}
fn decode_hex_digit(value: u8) -> Option<u8> {
match value {
b'0'..=b'9' => Some(value - b'0'),
b'a'..=b'f' => Some(value - b'a' + 10),
b'A'..=b'F' => Some(value - b'A' + 10),
_ => None,
}
}
fn escape_sql_string(value: &str) -> String {
value.replace('\\', "\\\\").replace('\'', "\\'")
}
type MigrateFuture<'e, T> = Pin<Box<dyn Future<Output = T> + Send + 'e>>;
impl<R> Migrate for ClickHouseMigrate<'_, R>
where
R: Fn(&str) -> String + Send + Sync,
{
fn create_schema_if_not_exists<'e>(
&'e mut self,
schema_name: &'e str,
) -> MigrateFuture<'e, Result<(), MigrateError>> {
Box::pin(async move {
if !valid_identifier(schema_name) {
return Err(migrate_error(Error::InvalidSchema));
}
let statement = format!("CREATE DATABASE IF NOT EXISTS `{schema_name}`");
execute_statement(self.client, self.connection, &statement, self.timeout)
.await
.map_err(migrate_error)
})
}
fn ensure_migrations_table<'e>(
&'e mut self,
table_name: &'e str,
) -> MigrateFuture<'e, Result<(), MigrateError>> {
Box::pin(async move {
let database = format!("`{}`", self.database);
execute_statement(
self.client,
self.connection,
&format!("CREATE DATABASE IF NOT EXISTS {database}"),
self.timeout,
)
.await
.map_err(migrate_error)?;
execute_statement(
self.client,
self.connection,
&format!(
"CREATE TABLE IF NOT EXISTS {database}.{table_name} \
(version Int64, description String, installed_on DateTime64(3) DEFAULT now64(3), \
success Bool, checksum String, execution_time Int64) ENGINE = MergeTree ORDER BY version"
),
self.timeout,
)
.await
.map_err(migrate_error)
})
}
fn dirty_version<'e>(
&'e mut self,
_table_name: &'e str,
) -> MigrateFuture<'e, Result<Option<i64>, MigrateError>> {
Box::pin(async { Ok(None) })
}
fn list_applied_migrations<'e>(
&'e mut self,
table_name: &'e str,
) -> MigrateFuture<'e, Result<Vec<AppliedMigration>, MigrateError>> {
Box::pin(async move {
let database = format!("`{}`", self.database);
let statement = format!(
"SELECT DISTINCT version, checksum FROM {database}.{table_name} \
WHERE success ORDER BY version FORMAT JSONEachRow"
);
let response = execute_sql(self.client, self.connection, &statement, self.timeout)
.await
.map_err(migrate_error)?;
response
.lines()
.filter(|line| !line.trim().is_empty())
.map(|line| {
let row = serde_json::from_str::<Applied>(line)
.map_err(|_| migrate_error(Error::InvalidResponse))?;
let checksum = decode_hex(&row.checksum).map_err(migrate_error)?;
Ok(AppliedMigration {
version: row.version,
checksum: checksum.into(),
})
})
.collect()
})
}
fn lock(&mut self) -> MigrateFuture<'_, Result<(), MigrateError>> {
Box::pin(async { Ok(()) })
}
fn unlock(&mut self) -> MigrateFuture<'_, Result<(), MigrateError>> {
Box::pin(async { Ok(()) })
}
fn apply<'e>(
&'e mut self,
table_name: &'e str,
migration: &'e Migration,
) -> MigrateFuture<'e, Result<Duration, MigrateError>> {
Box::pin(async move {
let started_at = Instant::now();
let statement = (self.render)(migration.sql.as_str());
execute_statement(self.client, self.connection, &statement, self.timeout)
.await
.map_err(|error| migrate_execution_error(error, migration.version))?;
let elapsed = started_at.elapsed();
let execution_time = elapsed.as_nanos().min(i64::MAX as u128) as i64;
let description = escape_sql_string(&migration.description);
let checksum = encode_hex(&migration.checksum);
let database = format!("`{}`", self.database);
execute_statement(
self.client,
self.connection,
&format!(
"INSERT INTO {database}.{table_name} \
(version, description, success, checksum, execution_time) \
VALUES ({}, '{}', true, '{}', {execution_time})",
migration.version, description, checksum
),
self.timeout,
)
.await
.map_err(migrate_error)?;
Ok(elapsed)
})
}
fn revert<'e>(
&'e mut self,
_table_name: &'e str,
_migration: &'e Migration,
) -> MigrateFuture<'e, Result<Duration, MigrateError>> {
Box::pin(async {
Err(MigrateError::Execute(SqlxError::AnyDriverError(Box::new(
std::io::Error::other("ClickHouse migrations are forward-only"),
))))
})
}
}
pub fn storage_error(error: &MigrateError) -> Option<&Error> {
let error = match error {
MigrateError::Execute(error) | MigrateError::ExecuteMigration(error, _) => error,
_ => return None,
};
match error {
SqlxError::AnyDriverError(error) => error.downcast_ref(),
_ => None,
}
}

View file

@ -110,6 +110,13 @@ pub async fn execute_read(
.body(sql.to_owned());
let mut response = request.send().await.map_err(|_| Error::Transport)?;
if !response.status().is_success() {
if response
.headers()
.get("x-clickhouse-exception-code")
.is_some_and(|code| code == "396")
{
return Err(Error::ResponseTooLarge);
}
return Err(Error::QueryFailed(response.status().as_u16()));
}

View file

@ -0,0 +1,476 @@
use std::time::Duration;
use litellm_http::Client;
use litellm_storage_clickhouse::{
ClickHouseMigrate, Connection, Error, READ_LIMITS, execute_statement, storage_error,
};
use rstest::{fixture, rstest};
use sqlx::{
SqlStr,
migrate::{Migrate, MigrateError, Migration, MigrationType, Migrator},
};
use testcontainers_modules::{
clickhouse::ClickHouse,
testcontainers::{ContainerAsync, ImageExt, runners::AsyncRunner},
};
use wiremock::{
Mock, MockServer, ResponseTemplate,
matchers::{body_string, method, query_param},
};
const CLICKHOUSE_TAG: &str =
"26.9.6.6@sha256:eb4870e7ca7ed70c259eebfcfbee6cf797017f6b5436c2926bbbfe3d4d28486e";
const DATABASE: &str = "storage_migrate_test";
const REQUEST_TIMEOUT: Duration = Duration::from_secs(10);
const SELECT_APPLIED: &str = "SELECT DISTINCT version, checksum FROM `trace_test`._sqlx_migrations \
WHERE success ORDER BY version FORMAT JSONEachRow";
type TestResult<T = ()> = Result<T, Box<dyn std::error::Error>>;
struct ClickHouseDatabase {
_container: ContainerAsync<ClickHouse>,
url: String,
client: Client,
}
#[fixture]
async fn database() -> TestResult<ClickHouseDatabase> {
let container = ClickHouse::default()
.with_tag(CLICKHOUSE_TAG)
.with_env_var("CLICKHOUSE_SKIP_USER_SETUP", "1")
.start()
.await?;
let url = format!(
"http://{}:{}",
container.get_host().await?,
container.get_host_port_ipv4(8123).await?
);
Ok(ClickHouseDatabase {
_container: container,
url,
client: Client::no_redirect_for_test(),
})
}
#[fixture]
async fn mock_server() -> MockServer {
let server = MockServer::start().await;
Mock::given(method("POST"))
.respond_with(ResponseTemplate::new(200))
.with_priority(10)
.mount(&server)
.await;
server
}
fn migration(version: i64, sql: &'static str) -> Migration {
Migration::new(
version,
format!("migration_{version}").into(),
MigrationType::Simple,
SqlStr::from_static(sql),
false,
)
}
fn migrator(migrations: Vec<Migration>) -> Migrator {
Migrator {
ignore_missing: true,
locking: false,
..Migrator::with_migrations(migrations)
}
}
fn render_database(sql: &str) -> String {
sql.replace("{database}", &format!("`{DATABASE}`"))
}
async fn run_migrations<R>(
database: &ClickHouseDatabase,
migrator: &Migrator,
schema: &str,
render: R,
) -> Result<(), MigrateError>
where
R: Fn(&str) -> String + Send + Sync,
{
let connection = Connection::writer(&database.url).expect("valid ClickHouse URL");
let mut adapter = ClickHouseMigrate::new(
&database.client,
&connection,
schema,
render,
REQUEST_TIMEOUT,
)
.expect("valid schema");
migrator.run_direct(None, &mut adapter, false).await
}
async fn execute_write(database: &ClickHouseDatabase, sql: &str) -> TestResult {
database
.client
.post(&database.url)
.body(sql.to_owned())
.send()
.await?
.error_for_status()?;
Ok(())
}
async fn read_json(database: &ClickHouseDatabase, sql: &str) -> TestResult<serde_json::Value> {
let response = database
.client
.post(&database.url)
.body(sql.to_owned())
.send()
.await?
.error_for_status()?;
Ok(serde_json::from_str(&response.text().await?)?)
}
async fn ledger_versions(database: &ClickHouseDatabase) -> TestResult<Vec<i64>> {
let response = read_json(
database,
&format!(
"SELECT version FROM `{DATABASE}`._sqlx_migrations \
GROUP BY version ORDER BY version FORMAT JSON"
),
)
.await?;
Ok(response["data"]
.as_array()
.expect("ClickHouse returns versions")
.iter()
.map(|row| row["version"].as_i64().expect("version is Int64"))
.collect())
}
fn encode_hex(bytes: &[u8]) -> String {
bytes
.iter()
.map(|byte| format!("{byte:02x}"))
.collect::<Vec<_>>()
.join("")
}
#[rstest]
#[tokio::test]
async fn only_pending_migrations_execute_on_the_second_run(
#[future(awt)] database: TestResult<ClickHouseDatabase>,
) -> TestResult {
let database = database?;
let migrator = migrator(vec![
migration(
1,
"CREATE TABLE IF NOT EXISTS {database}.migration_one (id UInt8) ENGINE = MergeTree ORDER BY id",
),
migration(
2,
"CREATE TABLE IF NOT EXISTS {database}.migration_two (id UInt8) ENGINE = MergeTree ORDER BY id",
),
]);
run_migrations(&database, &migrator, DATABASE, render_database).await?;
run_migrations(&database, &migrator, DATABASE, render_database).await?;
execute_statement(
&database.client,
&Connection::writer(&database.url)?,
"SYSTEM FLUSH LOGS",
REQUEST_TIMEOUT,
)
.await?;
let queries = read_json(
&database,
"SELECT count() AS executions FROM system.query_log \
WHERE type = 'QueryFinish' AND query LIKE \
'CREATE TABLE IF NOT EXISTS `storage_migrate_test`.migration_%' FORMAT JSON",
)
.await?;
assert_eq!(queries["data"][0]["executions"].as_u64(), Some(2));
assert_eq!(ledger_versions(&database).await?, vec![1, 2]);
Ok(())
}
#[rstest]
#[tokio::test]
async fn edited_migration_checksum_returns_version_mismatch(
#[future(awt)] database: TestResult<ClickHouseDatabase>,
) -> TestResult {
let database = database?;
let original = migrator(vec![migration(
1,
"CREATE TABLE IF NOT EXISTS {database}.original (id UInt8) ENGINE = MergeTree ORDER BY id",
)]);
let changed = migrator(vec![migration(
1,
"CREATE TABLE IF NOT EXISTS {database}.changed (id UInt8) ENGINE = MergeTree ORDER BY id",
)]);
run_migrations(&database, &original, DATABASE, render_database).await?;
assert!(matches!(
run_migrations(&database, &changed, DATABASE, render_database).await,
Err(MigrateError::VersionMismatch(1))
));
Ok(())
}
#[rstest]
#[tokio::test]
async fn failed_migration_is_not_recorded_and_retains_storage_error(
#[future(awt)] database: TestResult<ClickHouseDatabase>,
) -> TestResult {
let database = database?;
let migrator = migrator(vec![migration(1, "THIS IS NOT VALID CLICKHOUSE SQL")]);
let error = run_migrations(&database, &migrator, DATABASE, render_database)
.await
.expect_err("invalid SQL must fail");
assert!(matches!(&error, MigrateError::ExecuteMigration(_, 1)));
assert!(matches!(
storage_error(&error),
Some(Error::SchemaFailed(_))
));
let rows = read_json(
&database,
&format!(
"SELECT count() AS rows FROM `{DATABASE}`._sqlx_migrations \
WHERE version = 1 FORMAT JSON"
),
)
.await?;
assert_eq!(rows["data"][0]["rows"].as_u64(), Some(0));
Ok(())
}
#[rstest]
fn invalid_database_identifier_is_rejected() {
let client = Client::no_redirect_for_test();
let connection = Connection::writer("http://127.0.0.1:1").expect("valid URL");
assert!(matches!(
ClickHouseMigrate::new(
&client,
&connection,
"storage_test; DROP DATABASE default",
str::to_owned,
REQUEST_TIMEOUT,
),
Err(Error::InvalidSchema)
));
}
#[rstest]
#[tokio::test]
async fn unknown_source_version_is_tolerated(
#[future(awt)] database: TestResult<ClickHouseDatabase>,
) -> TestResult {
let database = database?;
run_migrations(&database, &migrator(vec![]), DATABASE, render_database).await?;
execute_write(
&database,
&format!(
"INSERT INTO `{DATABASE}`._sqlx_migrations \
(version, description, success, checksum, execution_time) \
VALUES (99, 'unknown', true, '{}', 0)",
"00".repeat(48)
),
)
.await?;
let migrator = migrator(vec![migration(
1,
"CREATE TABLE IF NOT EXISTS {database}.known (id UInt8) ENGINE = MergeTree ORDER BY id",
)]);
run_migrations(&database, &migrator, DATABASE, render_database).await?;
assert_eq!(ledger_versions(&database).await?, vec![1, 99]);
Ok(())
}
#[rstest]
#[tokio::test]
async fn duplicate_ledger_rows_are_tolerated(
#[future(awt)] database: TestResult<ClickHouseDatabase>,
) -> TestResult {
let database = database?;
let applied = migration(
1,
"CREATE TABLE IF NOT EXISTS {database}.duplicate_test (id UInt8) ENGINE = MergeTree ORDER BY id",
);
let checksum = encode_hex(&applied.checksum);
let migrator = migrator(vec![applied]);
run_migrations(&database, &migrator, DATABASE, render_database).await?;
execute_write(
&database,
&format!(
"INSERT INTO `{DATABASE}`._sqlx_migrations \
(version, description, success, checksum, execution_time) \
VALUES (1, 'migration_1', true, '{checksum}', 0)"
),
)
.await?;
run_migrations(&database, &migrator, DATABASE, render_database).await?;
let rows = read_json(
&database,
&format!(
"SELECT count() AS rows, uniqExact(version) AS versions \
FROM `{DATABASE}`._sqlx_migrations FORMAT JSON"
),
)
.await?;
assert_eq!(rows["data"][0]["rows"].as_u64(), Some(2));
assert_eq!(rows["data"][0]["versions"].as_u64(), Some(1));
Ok(())
}
#[rstest]
#[tokio::test]
async fn revert_reports_forward_only_error() {
let client = Client::no_redirect_for_test();
let connection = Connection::writer("http://127.0.0.1:1").expect("valid URL");
let migration = migration(
1,
"CREATE TABLE IF NOT EXISTS {database}.revert_test (id UInt8) ENGINE = MergeTree ORDER BY id",
);
let mut adapter = ClickHouseMigrate::new(
&client,
&connection,
DATABASE,
render_database,
REQUEST_TIMEOUT,
)
.expect("valid schema");
let error = adapter
.revert("_sqlx_migrations", &migration)
.await
.expect_err("ClickHouse migrations cannot be reverted");
assert!(
error
.to_string()
.contains("ClickHouse migrations are forward-only")
);
}
#[rstest]
#[case::numeric("1")]
#[case::quoted("\"1\"")]
#[tokio::test]
async fn applied_int64_versions_accept_numeric_and_quoted_json(
#[future(awt)] mock_server: MockServer,
#[case] version: &str,
) {
Mock::given(method("POST"))
.and(body_string(SELECT_APPLIED))
.respond_with(
ResponseTemplate::new(200)
.set_body_string(format!("{{\"version\":{version},\"checksum\":\"00\"}}\n")),
)
.mount(&mock_server)
.await;
let client = Client::no_redirect_for_test();
let connection = Connection::writer(&mock_server.uri()).expect("valid URL");
let migrator = migrator(vec![]);
let mut adapter = ClickHouseMigrate::new(
&client,
&connection,
"trace_test",
str::to_owned,
REQUEST_TIMEOUT,
)
.expect("valid schema");
migrator
.run_direct(None, &mut adapter, false)
.await
.expect("applied version parses");
}
#[rstest]
#[tokio::test]
async fn schema_requests_override_unsafe_connection_settings(
#[future(awt)] mock_server: MockServer,
) {
Mock::given(method("POST"))
.and(query_param("wait_end_of_query", "1"))
.and(query_param("send_progress_in_http_headers", "0"))
.and(query_param("async_insert", "0"))
.and(query_param("custom_setting", "preserved"))
.respond_with(ResponseTemplate::new(200))
.expect(1)
.mount(&mock_server)
.await;
let url = format!(
"{}/?wait_end_of_query=0&send_progress_in_http_headers=1&async_insert=1&custom_setting=preserved",
mock_server.uri()
);
execute_statement(
&Client::no_redirect_for_test(),
&Connection::writer(&url).expect("valid URL"),
"CREATE DATABASE IF NOT EXISTS trace_test",
REQUEST_TIMEOUT,
)
.await
.expect("schema execution succeeds");
let requests = mock_server
.received_requests()
.await
.expect("requests recorded");
for name in [
"wait_end_of_query",
"send_progress_in_http_headers",
"async_insert",
] {
assert_eq!(
requests[0]
.url
.query_pairs()
.filter(|(key, _)| key == name)
.count(),
1
);
}
}
#[rstest]
#[tokio::test]
async fn oversized_ledger_response_is_rejected_before_migrations(
#[future(awt)] mock_server: MockServer,
) {
Mock::given(method("POST"))
.and(body_string(SELECT_APPLIED))
.respond_with(
ResponseTemplate::new(200).set_body_string(" ".repeat(READ_LIMITS.response_bytes + 1)),
)
.mount(&mock_server)
.await;
let client = Client::no_redirect_for_test();
let connection = Connection::writer(&mock_server.uri()).expect("valid URL");
let migrator = migrator(vec![]);
let mut adapter = ClickHouseMigrate::new(
&client,
&connection,
"trace_test",
str::to_owned,
REQUEST_TIMEOUT,
)
.expect("valid schema");
let error = migrator
.run_direct(None, &mut adapter, false)
.await
.expect_err("oversized result is rejected");
assert!(matches!(
storage_error(&error),
Some(Error::ResponseTooLarge)
));
assert_eq!(
mock_server
.received_requests()
.await
.expect("requests recorded")
.len(),
3
);
}

View file

@ -111,3 +111,83 @@ async fn typed_fetch_encodes_parameters_and_validates_rows(
assert!(matches!(envelope, Err(Error::InvalidResponse)));
}
}
#[rstest]
#[case::result_limit("396", true)]
#[case::memory_limit("241", false)]
#[case::timeout("159", false)]
#[case::unknown("", false)]
#[tokio::test]
async fn server_result_limits_allow_smaller_pages_without_retrying_other_failures(
#[case] code: &str,
#[case] result_limit: bool,
) {
use wiremock::{Mock, MockServer, ResponseTemplate, matchers::method};
let server = MockServer::start().await;
Mock::given(method("POST"))
.respond_with(ResponseTemplate::new(500).insert_header("X-ClickHouse-Exception-Code", code))
.expect(1)
.mount(&server)
.await;
let connection = Connection::parse(&server.uri()).unwrap();
let error = execute_read(
&Client::no_redirect_for_test(),
&connection,
"SELECT 1",
&BTreeMap::new(),
)
.await
.unwrap_err();
if result_limit {
assert!(matches!(error, Error::ResponseTooLarge));
} else {
assert!(matches!(error, Error::QueryFailed(500)));
}
}
#[test]
fn insert_timeout_environment_controls_transport() {
for value in ["1", "3", "0", "invalid"] {
let result = std::process::Command::new(std::env::current_exe().unwrap())
.args(["--exact", "insert_timeout_environment_child"])
.env("LITELLM_TEST_INSERT_TIMEOUT", value)
.env("CLICKHOUSE_INSERT_TIMEOUT_SECONDS", value)
.output()
.unwrap();
assert!(
result.status.success(),
"{}",
String::from_utf8_lossy(&result.stdout)
);
}
}
#[tokio::test]
async fn insert_timeout_environment_child() {
use wiremock::{Mock, MockServer, ResponseTemplate, matchers::method};
let Ok(value) = std::env::var("LITELLM_TEST_INSERT_TIMEOUT") else {
return;
};
let server = MockServer::start().await;
Mock::given(method("POST"))
.respond_with(ResponseTemplate::new(200).set_delay(std::time::Duration::from_millis(1500)))
.mount(&server)
.await;
let result = insert_encoded_rows(
&Client::no_redirect_for_test(),
&Connection::parse(&server.uri()).unwrap(),
"traces",
"otel_traces",
"token",
"{}",
)
.await;
match value.as_str() {
"1" => assert!(matches!(result, Err(Error::Transport))),
"3" => assert!(result.is_ok()),
_ => assert!(matches!(
result,
Err(Error::InvalidLimit("CLICKHOUSE_INSERT_TIMEOUT_SECONDS"))
)),
}
}

View file

@ -0,0 +1,5 @@
Own storage-independent trace reads over `TraceStore`: the in-process read cache (identity, single-flight, freshness expiry, weighting), cursor formats, paging, response splitting, spend windows and run batching
Depend on trace domain types, never storage, HTTP or Python
Preserve the full source and authorization scope in every cache key
Keep snapshots immutable and expose borrowed data
Storage adapters implement `TraceStore`; keep SQL and row encoding there

View file

@ -0,0 +1,21 @@
[package]
name = "litellm-traces-cache"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
base64.workspace = true
litellm-traces.workspace = true
moka.workspace = true
serde.workspace = true
serde_json.workspace = true
sha2.workspace = true
time.workspace = true
thiserror.workspace = true
tracing.workspace = true
[dev-dependencies]
rstest.workspace = true
tokio.workspace = true

View file

@ -0,0 +1,395 @@
use std::{future::Future, sync::Arc, time::Duration};
use litellm_traces::{
Trace, TraceSummary,
query::named::{ReadAccessParams, TraceSpansRow},
};
use moka::{Expiry, future::Cache};
use serde::Serialize;
use sha2::{Digest, Sha256};
use crate::Error;
pub const LIVE_TTL: Duration = Duration::from_secs(5);
pub const SETTLED_TTL: Duration = Duration::from_secs(10 * 60);
const SETTLED_AFTER_MS: u64 = 5 * 60 * 1000;
const MAX_INDEX_ENTRIES: u64 = 100_000;
#[derive(Clone, Eq, Hash, PartialEq)]
pub struct SnapshotKey(String);
impl SnapshotKey {
fn digest(fields: &impl Serialize) -> Result<Self, Error> {
Ok(Self(format!(
"{:x}",
Sha256::digest(serde_json::to_vec(fields)?)
)))
}
pub fn new(
source: &str,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
snapshot_ms: u64,
) -> Result<Self, Error> {
Self::digest(&(source, access, trace_id, trace_ref, snapshot_ms))
}
pub fn latest(
source: &str,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
) -> Result<Self, Error> {
Self::digest(&(source, access, trace_id, trace_ref))
}
pub(crate) fn run(
source: &str,
access: &ReadAccessParams,
run: (&str, &str, &str, &str),
) -> Result<Self, Error> {
Self::digest(&("run", source, access, run))
}
pub(crate) fn scope(source: &str, access: &ReadAccessParams) -> Result<Self, Error> {
Self::digest(&("scope", source, access))
}
}
/// How long a read result stays reusable: traces still receiving spans, or read with spend
/// unavailable, are re-read after `LIVE_TTL`; traces quiet for `SETTLED_AFTER_MS` are kept for
/// `SETTLED_TTL`.
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum Freshness {
Live,
Settled,
}
impl Freshness {
pub fn of(rows: &[TraceSpansRow], spend_known: bool, snapshot_ms: u64) -> Self {
let last_end_ms = rows
.iter()
.map(|row| row.start_ns.saturating_add_unsigned(row.duration_ns) / 1_000_000)
.max()
.unwrap_or(i64::MAX);
let quiet_ms = i64::try_from(snapshot_ms)
.unwrap_or(i64::MAX)
.saturating_sub(last_end_ms);
if spend_known && quiet_ms >= SETTLED_AFTER_MS as i64 {
Self::Settled
} else {
Self::Live
}
}
fn ttl(self) -> Duration {
match self {
Self::Live => LIVE_TTL,
Self::Settled => SETTLED_TTL,
}
}
}
trait Fresh {
fn freshness(&self) -> Freshness;
}
struct ByFreshness;
impl<K, V: Fresh> Expiry<K, V> for ByFreshness {
fn expire_after_create(&self, _: &K, value: &V, _: std::time::Instant) -> Option<Duration> {
Some(value.freshness().ttl())
}
}
pub struct Snapshot {
trace: Trace,
version: String,
snapshot_ms: u64,
freshness: Freshness,
weight: u32,
}
impl Snapshot {
pub fn trace(&self) -> &Trace {
&self.trace
}
pub fn version(&self) -> &str {
&self.version
}
pub fn snapshot_ms(&self) -> u64 {
self.snapshot_ms
}
pub fn freshness(&self) -> Freshness {
self.freshness
}
}
#[derive(Clone, Copy)]
struct Latest {
snapshot_ms: u64,
freshness: Freshness,
}
impl Fresh for Latest {
fn freshness(&self) -> Freshness {
self.freshness
}
}
/// Resolved trace snapshots pinned by `snapshot_ms` for paging, plus which snapshot each trace
/// currently serves so repeated opens reuse one read until its freshness expires.
pub struct SnapshotCache {
pinned: Cache<SnapshotKey, Arc<Snapshot>>,
latest: Cache<SnapshotKey, Latest>,
max_graph_bytes: usize,
}
impl SnapshotCache {
pub fn new(max_graph_bytes: usize, idle: Duration) -> Self {
Self {
pinned: Cache::builder()
.max_capacity((max_graph_bytes as u64).saturating_mul(2))
.weigher(|_: &SnapshotKey, snapshot: &Arc<Snapshot>| snapshot.weight)
.time_to_idle(idle)
.build(),
latest: Cache::builder()
.max_capacity(MAX_INDEX_ENTRIES)
.expire_after(ByFreshness)
.build(),
max_graph_bytes,
}
}
pub async fn get(&self, key: &SnapshotKey) -> Option<Arc<Snapshot>> {
self.pinned.get(key).await
}
/// Returns the snapshot pinned at `key`, running `load` once for all concurrent callers on a
/// miss. A failed load is not cached.
pub async fn pinned_or_load<E, F>(
&self,
key: SnapshotKey,
snapshot_ms: u64,
load: F,
) -> Result<Arc<Snapshot>, Arc<E>>
where
E: From<Error> + Send + Sync + 'static,
F: Future<Output = Result<(Trace, Freshness), E>>,
{
self.pinned
.try_get_with(key, async {
let (trace, freshness) = load.await?;
Ok(Arc::new(self.snapshot(trace, snapshot_ms, freshness)?))
})
.await
}
/// Returns the snapshot `latest` currently serves. On a miss, `load_at(now_ms)` runs once for
/// all concurrent callers and its snapshot is served until its freshness expires.
pub async fn latest_or_load<E, F, Fut>(
&self,
latest: SnapshotKey,
now_ms: u64,
load_at: F,
) -> Result<Arc<Snapshot>, Arc<E>>
where
E: Send + Sync + 'static,
F: Fn(u64) -> Fut,
Fut: Future<Output = Result<Arc<Snapshot>, Arc<E>>>,
{
let entry = self
.latest
.try_get_with(latest, async {
let snapshot = load_at(now_ms).await?;
Ok::<_, Arc<E>>(Latest {
snapshot_ms: snapshot.snapshot_ms,
freshness: snapshot.freshness,
})
})
.await
.map_err(|error| Arc::clone(&*error))?;
load_at(entry.snapshot_ms).await
}
fn snapshot(
&self,
trace: Trace,
snapshot_ms: u64,
freshness: Freshness,
) -> Result<Snapshot, Error> {
let encoded = serde_json::to_vec(&trace)?;
if encoded.len() > self.max_graph_bytes {
return Err(Error::ReadTooLarge);
}
let span_ids: Vec<&str> = trace
.spans
.iter()
.map(|span| span.span_id.as_str())
.collect();
let version = format!("{:x}", Sha256::digest(serde_json::to_vec(&span_ids)?));
Ok(Snapshot {
trace,
version,
snapshot_ms,
freshness,
weight: u32::try_from(encoded.len().saturating_mul(2)).unwrap_or(u32::MAX),
})
}
#[cfg(test)]
async fn weighted_size(&self) -> u64 {
self.pinned.run_pending_tasks().await;
self.pinned.weighted_size()
}
}
#[derive(Clone)]
pub(crate) enum ListedRun {
Resolved(Box<TraceSummary>, Freshness),
Limited,
}
impl Fresh for ListedRun {
fn freshness(&self) -> Freshness {
match self {
Self::Resolved(_, freshness) => *freshness,
Self::Limited => Freshness::Settled,
}
}
}
pub(crate) struct ListCache {
pub(crate) runs: Cache<SnapshotKey, ListedRun>,
pub(crate) limits: Cache<SnapshotKey, u32>,
}
impl ListCache {
pub(crate) fn new() -> Self {
Self {
runs: Cache::builder()
.max_capacity(MAX_INDEX_ENTRIES)
.expire_after(ByFreshness)
.build(),
limits: Cache::builder()
.max_capacity(MAX_INDEX_ENTRIES)
.time_to_live(SETTLED_TTL)
.build(),
}
}
}
#[cfg(test)]
mod tests {
use litellm_traces::{
SpanStatus,
query::named::{SpendByResponseIdsRow, TraceSpansRow},
resolve_trace,
};
use rstest::rstest;
use super::*;
fn row(span_id: &str) -> TraceSpansRow {
TraceSpansRow {
trace_id: String::new(),
span_id: span_id.into(),
parent_span_id: String::new(),
name: "run".into(),
kind: litellm_traces::ObservationType::Agent,
wrapper_candidate: false,
agent: "agent".into(),
framework: String::new(),
status: SpanStatus::Ok,
status_message: String::new(),
error_truncated: false,
start_ns: 1_790_742_989_000_000_000,
duration_ns: 10_000_000,
service: "agent-demo".into(),
input_preview: format!("input of {span_id}"),
model: String::new(),
input_tokens: 0,
output_tokens: 0,
litellm_request_id: String::new(),
call_keys: Vec::new(),
call_evidence: None,
tool_call_id: String::new(),
team_id: String::new(),
api_key_hash: String::new(),
user_id: String::new(),
}
}
fn trace(span_id: &str) -> Trace {
resolve_trace(
"trace",
"ref",
&[row(span_id)],
&[] as &[SpendByResponseIdsRow],
)
.expect("fixture should resolve")
}
fn key(suffix: &str) -> SnapshotKey {
SnapshotKey::new(
"source",
&ReadAccessParams {
all_teams: false,
user_id: String::new(),
team_ids: vec!["team".into()],
},
suffix,
"ref",
100,
)
.unwrap()
}
#[tokio::test]
async fn weighted_capacity_bounds_retained_snapshots() {
let limit = ["first", "second", "third"]
.iter()
.map(|span_id| serde_json::to_vec(&trace(span_id)).unwrap().len())
.max()
.unwrap();
let cache = SnapshotCache::new(limit, Duration::from_secs(120));
for (key, span_id) in [
(key("a"), "first"),
(key("b"), "second"),
(key("c"), "third"),
] {
cache
.pinned_or_load(key, 100, async {
Ok::<_, Error>((trace(span_id), Freshness::Settled))
})
.await
.unwrap();
}
assert!(cache.weighted_size().await <= (limit as u64) * 2);
}
const LAST_END_MS: u64 = 1_790_742_989_010;
#[rstest]
#[case::just_ended(LAST_END_MS, true, Freshness::Live)]
#[case::quiet_just_under(LAST_END_MS + SETTLED_AFTER_MS - 1, true, Freshness::Live)]
#[case::quiet_long_enough(LAST_END_MS + SETTLED_AFTER_MS, true, Freshness::Settled)]
#[case::spend_unknown(LAST_END_MS + SETTLED_AFTER_MS, false, Freshness::Live)]
fn freshness_settles_once_spans_stop_and_spend_is_known(
#[case] snapshot_ms: u64,
#[case] spend_known: bool,
#[case] expected: Freshness,
) {
assert_eq!(
Freshness::of(&[row("root")], spend_known, snapshot_ms),
expected
);
}
}

View file

@ -0,0 +1,124 @@
use base64::{Engine, engine::general_purpose::URL_SAFE};
use serde::{Deserialize, Serialize};
use crate::ReadError;
pub(super) fn encode_cursor<T: Serialize>(position: &T) -> String {
URL_SAFE.encode(serde_json::to_vec(position).unwrap_or_default())
}
pub(super) fn decode_cursor<T: for<'de> Deserialize<'de>, E>(
cursor: &str,
kind: &'static str,
) -> Result<T, ReadError<E>> {
URL_SAFE
.decode(cursor)
.ok()
.and_then(|json| serde_json::from_slice(&json).ok())
.ok_or(ReadError::InvalidCursor(kind))
}
pub(super) fn trace_position<E>(cursor: Option<&str>) -> Result<(i64, String), ReadError<E>> {
let Some(cursor) = cursor.filter(|cursor| !cursor.is_empty()) else {
return Ok((0, String::new()));
};
match decode_cursor::<(i64, String), E>(cursor, "trace")? {
(start_ms, trace_ref) if start_ms > 0 && !trace_ref.is_empty() => Ok((start_ms, trace_ref)),
_ => Err(ReadError::InvalidCursor("trace")),
}
}
#[derive(Deserialize, Serialize)]
pub(super) struct ErrorPosition {
pub(super) offset: u64,
pub(super) version: String,
}
pub(super) fn error_position<E>(
cursor: Option<&str>,
) -> Result<Option<ErrorPosition>, ReadError<E>> {
let Some(cursor) = cursor else {
return Ok(None);
};
let position: ErrorPosition = decode_cursor(cursor, "diagnostic")?;
let valid_version = position.version.len() == 64
&& position
.version
.bytes()
.all(|byte| byte.is_ascii_digit() || (b'A'..=b'F').contains(&byte));
if i64::try_from(position.offset).is_err() || !valid_version {
return Err(ReadError::InvalidCursor("diagnostic"));
}
Ok(Some(position))
}
#[derive(Deserialize, Serialize)]
pub(super) struct SpanPosition {
pub(super) trace_ref: String,
pub(super) snapshot_ms: u64,
pub(super) offset: usize,
pub(super) version: String,
}
#[cfg(test)]
mod tests {
use rstest::rstest;
use super::*;
#[rstest]
fn trace_cursor_round_trips_the_last_listed_run() {
let cursor = encode_cursor(&(1_790_742_989_377_i64, "4bad42b84e9de3ba46fc870185f8f023"));
assert_eq!(
trace_position::<std::io::Error>(Some(&cursor)).unwrap(),
(
1_790_742_989_377,
"4bad42b84e9de3ba46fc870185f8f023".to_owned()
)
);
assert_eq!(
trace_position::<std::io::Error>(None).unwrap(),
(0, String::new())
);
assert_eq!(
trace_position::<std::io::Error>(Some("")).unwrap(),
(0, String::new())
);
}
#[rstest]
#[case::not_base64("abc")]
#[case::not_json("bm90LWpzb24=")]
#[case::numeric_reference("WzEsIDJd")]
#[case::zero_start("WzAsICJ0Il0=")]
fn malformed_trace_cursors_are_rejected(#[case] cursor: &str) {
let result: Result<(i64, String), ReadError<std::io::Error>> = trace_position(Some(cursor));
assert!(matches!(result, Err(ReadError::InvalidCursor("trace"))));
}
#[rstest]
#[case::not_base64("garbage")]
#[case::missing_fields("e30=")]
#[case::not_an_object("WzEsMl0=")]
fn malformed_diagnostic_cursors_are_rejected(#[case] cursor: &str) {
let result: Result<Option<ErrorPosition>, ReadError<std::io::Error>> =
error_position(Some(cursor));
assert!(matches!(
result,
Err(ReadError::InvalidCursor("diagnostic"))
));
}
#[rstest]
#[case::lowercase_version("a".repeat(64))]
#[case::short_version("A".repeat(63))]
fn diagnostic_cursor_requires_a_content_version(#[case] version: String) {
let cursor = encode_cursor(&ErrorPosition { offset: 1, version });
let result: Result<Option<ErrorPosition>, ReadError<std::io::Error>> =
error_position(Some(&cursor));
assert!(matches!(
result,
Err(ReadError::InvalidCursor("diagnostic"))
));
}
}

View file

@ -0,0 +1,51 @@
use std::sync::Arc;
#[derive(Debug, thiserror::Error)]
pub enum Error {
#[error("trace snapshot serialization failed")]
Serialization(#[from] serde_json::Error),
#[error("trace snapshot exceeds the size limit")]
ReadTooLarge,
}
/// Cheap to clone so one failed single-flight read can be returned to every waiting caller.
#[derive(Debug, thiserror::Error)]
pub enum ReadError<E> {
#[error("invalid trace read parameters")]
InvalidParameters,
#[error("Invalid {0} cursor")]
InvalidCursor(&'static str),
#[error("Multiple traces have this ID; provide trace_ref")]
AmbiguousTrace,
#[error("Trace changed while paging; refresh the trace to continue")]
TraceChanged,
#[error("Trace exceeds the interactive read budget; use a filtered trace query")]
TooLarge,
#[error("trace could not be encoded")]
Encode(#[source] Arc<serde_json::Error>),
#[error(transparent)]
Store(Arc<E>),
}
impl<E> Clone for ReadError<E> {
fn clone(&self) -> Self {
match self {
Self::InvalidParameters => Self::InvalidParameters,
Self::InvalidCursor(kind) => Self::InvalidCursor(kind),
Self::AmbiguousTrace => Self::AmbiguousTrace,
Self::TraceChanged => Self::TraceChanged,
Self::TooLarge => Self::TooLarge,
Self::Encode(error) => Self::Encode(Arc::clone(error)),
Self::Store(error) => Self::Store(Arc::clone(error)),
}
}
}
impl<E> From<Error> for ReadError<E> {
fn from(error: Error) -> Self {
match error {
Error::ReadTooLarge => Self::TooLarge,
Error::Serialization(error) => Self::Encode(Arc::new(error)),
}
}
}

View file

@ -0,0 +1,12 @@
mod cache;
mod cursor;
mod error;
mod list;
mod reader;
mod spend;
mod store;
pub use cache::{Freshness, LIVE_TTL, SETTLED_TTL, Snapshot, SnapshotCache, SnapshotKey};
pub use error::{Error, ReadError};
pub use reader::{MAX_GRAPH_BYTES, MAX_GRAPH_SPANS, TraceReader};
pub use store::{StoreError, TraceStore};

View file

@ -0,0 +1,197 @@
use std::collections::HashMap;
use crate::{
ReadError, SnapshotKey, TraceReader, TraceStore,
cache::{Freshness, ListedRun},
reader::{map_store_error, now_ms},
spend::{spend, spend_window, spend_within},
store::StoreError,
};
use litellm_traces::{
TraceSummary, listed_summary,
query::named::{ListTracesRow, ReadAccessParams, TracePageSpansParams, TraceSpansRow},
resolve_trace,
};
const RUNS_PER_SPAN_READ: usize = 16;
fn run_key(team_id: &str, api_key_hash: &str, trace_id: &str) -> (String, String, String) {
(
team_id.to_owned(),
api_key_hash.to_owned(),
trace_id.to_owned(),
)
}
fn cache_key<E>(
source: &str,
access: &ReadAccessParams,
row: &ListTracesRow,
) -> Result<SnapshotKey, ReadError<E>> {
Ok(SnapshotKey::run(
source,
access,
(
&row.team_id,
&row.api_key_hash,
&row.trace_id,
&row.trace_ref,
),
)?)
}
fn summary(row: &ListTracesRow, listed: Option<&ListedRun>) -> TraceSummary {
match listed {
Some(ListedRun::Resolved(summary, _)) => (**summary).clone(),
Some(ListedRun::Limited) | None => listed_summary(row),
}
}
/// Summaries for one batch of listed runs. Runs resolved within their freshness window come
/// from the cache; only the rest are read from storage, with one span and one spend read.
pub(super) async fn list_summaries<S: TraceStore>(
reader: &TraceReader,
store: &S,
access: &ReadAccessParams,
runs: &[ListTracesRow],
) -> Result<Vec<TraceSummary>, ReadError<S::Error>> {
let mut keys = Vec::with_capacity(runs.len());
let mut listed = Vec::with_capacity(runs.len());
for row in runs {
let key = cache_key(store.source(), access, row)?;
listed.push(reader.lists.runs.get(&key).await);
keys.push(key);
}
let misses: Vec<&ListTracesRow> = runs
.iter()
.zip(&listed)
.filter_map(|(row, listed)| listed.is_none().then_some(row))
.collect();
let mut resolved = resolve_runs(reader, store, access, &misses)
.await?
.into_iter();
let mut summaries = Vec::with_capacity(runs.len());
for ((row, key), cached) in runs.iter().zip(keys).zip(listed) {
let listed = match cached {
Some(listed) => Some(listed),
None => {
let listed = resolved.next().flatten();
if let Some(listed) = &listed {
reader.lists.runs.insert(key, listed.clone()).await;
}
listed
}
};
summaries.push(summary(row, listed.as_ref()));
}
Ok(summaries)
}
/// One entry per run; `None` means the run could not be resolved and keeps its listed summary
/// without being cached.
async fn resolve_runs<S: TraceStore>(
reader: &TraceReader,
store: &S,
access: &ReadAccessParams,
runs: &[&ListTracesRow],
) -> Result<Vec<Option<ListedRun>>, ReadError<S::Error>> {
let (Some(start_ms), Some(end_ms)) = (
runs.iter().map(|row| row.start_ms).min(),
runs.iter()
.map(|row| row.start_ms.saturating_add(row.duration_ms))
.max(),
) else {
return Ok(Vec::new());
};
let params = TracePageSpansParams {
access: access.clone(),
trace_refs: runs.iter().map(|row| row.trace_ref.clone()).collect(),
start_ms,
end_ms: end_ms.saturating_add(1),
};
let snapshot_ms = now_ms();
let spans = match store.run_spans(&params, snapshot_ms).await {
Ok(spans) => spans,
Err(StoreError::TooLarge) => {
let mut resolved = Vec::with_capacity(runs.len());
for row in runs {
resolved.push(resolve_run(reader, store, access, row).await?);
}
return Ok(resolved);
}
Err(error) => return Err(map_store_error(error)),
};
let Some(spend_rows) = spend(store, access, &spans).await else {
// The batch's combined spend read failed; a run's own narrower window may still
// resolve, so fall back per run instead of leaving every run in the batch costless.
let mut resolved = Vec::with_capacity(runs.len());
for row in runs {
resolved.push(resolve_run(reader, store, access, row).await?);
}
return Ok(resolved);
};
let mut spans = spans;
spans.sort_by(|left, right| {
run_key(&left.team_id, &left.api_key_hash, &left.trace_id)
.cmp(&run_key(
&right.team_id,
&right.api_key_hash,
&right.trace_id,
))
.then(left.start_ns.cmp(&right.start_ns))
});
let by_run: HashMap<_, &[TraceSpansRow]> = spans
.chunk_by(|left, right| {
(&left.team_id, &left.api_key_hash, &left.trace_id)
== (&right.team_id, &right.api_key_hash, &right.trace_id)
})
.map(|run| {
(
run_key(&run[0].team_id, &run[0].api_key_hash, &run[0].trace_id),
run,
)
})
.collect();
Ok(runs
.iter()
.map(|row| {
let spans = by_run
.get(&run_key(&row.team_id, &row.api_key_hash, &row.trace_id))
.copied()
.unwrap_or_default();
let spend =
spend_window(spans).map_or(&[][..], |window| spend_within(&spend_rows, window));
resolve_trace(&row.trace_id, &row.trace_ref, spans, spend).map(|trace| {
ListedRun::Resolved(
Box::new(trace.summary),
Freshness::of(spans, true, snapshot_ms),
)
})
})
.collect())
}
async fn resolve_run<S: TraceStore>(
reader: &TraceReader,
store: &S,
access: &ReadAccessParams,
row: &ListTracesRow,
) -> Result<Option<ListedRun>, ReadError<S::Error>> {
match reader
.current(store, access, &row.trace_id, &row.trace_ref)
.await
{
Ok(snapshot) => Ok(snapshot.map(|snapshot| {
ListedRun::Resolved(
Box::new(snapshot.trace().summary.clone()),
snapshot.freshness(),
)
})),
Err(ReadError::TooLarge) => Ok(Some(ListedRun::Limited)),
Err(error) => Err(error),
}
}
pub(super) fn run_batches<T>(runs: &[T]) -> impl Iterator<Item = &[T]> + '_ {
runs.chunks(RUNS_PER_SPAN_READ)
}

View file

@ -0,0 +1,365 @@
use std::{sync::Arc, time::Duration};
use crate::{
ReadError, Snapshot, SnapshotCache, SnapshotKey, StoreError, TraceStore,
cache::{Freshness, ListCache},
cursor::{
ErrorPosition, SpanPosition, decode_cursor, encode_cursor, error_position, trace_position,
},
list::{list_summaries, run_batches},
spend::spend,
};
use litellm_traces::{
SpanDetail, SpanErrorPage, Trace, TracePage,
query::named::{
ListTracesParams, ReadAccessParams, SpanDetailParams, SpanErrorParams, TraceIdentityParams,
TraceSpansParams,
},
resolve_trace, to_ui_content,
};
pub const MAX_GRAPH_BYTES: usize = 64 * 1024 * 1024;
pub const MAX_GRAPH_SPANS: usize = 100_000;
const SNAPSHOT_IDLE: Duration = Duration::from_secs(120);
/// A read that found no trace, kept apart from failures so single-flight waiters share it
/// without it being cached.
pub(super) enum Miss<E> {
Absent,
Read(ReadError<E>),
}
impl<E> From<crate::Error> for Miss<E> {
fn from(error: crate::Error) -> Self {
Self::Read(error.into())
}
}
fn settle<T, E>(result: Result<T, Arc<Miss<E>>>) -> Result<Option<T>, ReadError<E>> {
match result {
Ok(value) => Ok(Some(value)),
Err(miss) => match &*miss {
Miss::Absent => Ok(None),
Miss::Read(error) => Err(error.clone()),
},
}
}
pub struct TraceReader {
snapshots: SnapshotCache,
pub(super) lists: ListCache,
response_bytes: usize,
}
impl TraceReader {
pub fn new(response_bytes: usize) -> Self {
Self {
snapshots: SnapshotCache::new(MAX_GRAPH_BYTES, SNAPSHOT_IDLE),
lists: ListCache::new(),
response_bytes,
}
}
pub async fn list_traces<S: TraceStore>(
&self,
store: &S,
access: &ReadAccessParams,
start_ms: i64,
end_ms: i64,
cursor: Option<&str>,
limit: u32,
) -> Result<TracePage, ReadError<S::Error>> {
if limit == 0 {
return Err(ReadError::InvalidParameters);
}
let (cursor_ms, cursor_trace_id) = trace_position(cursor)?;
let scope = SnapshotKey::scope(store.source(), access)?;
let accepted = self.lists.limits.get(&scope).await.unwrap_or(u32::MAX);
let mut params = ListTracesParams {
access: access.clone(),
start_ms,
end_ms,
cursor_ms,
cursor_trace_id,
limit: limit.min(500).min(accepted),
};
let page = loop {
match store.list_runs(&params).await {
Err(StoreError::TooLarge) if params.limit > 1 => {
params.limit /= 2;
self.lists.limits.insert(scope.clone(), params.limit).await;
}
Err(StoreError::TooLarge) => return Err(ReadError::TooLarge),
result => break result.map_err(map_store_error)?,
}
};
let next_cursor = page
.last()
.filter(|_| page.len() == params.limit as usize)
.map(|last| encode_cursor(&(last.start_ms, &last.trace_ref)));
let data = {
let mut summaries = Vec::with_capacity(page.len());
for batch in run_batches(&page) {
summaries.extend(list_summaries(self, store, access, batch).await?);
}
summaries
};
Ok(TracePage { data, next_cursor })
}
pub async fn get_trace<S: TraceStore>(
&self,
store: &S,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
) -> Result<Option<Trace>, ReadError<S::Error>> {
let Some(trace_ref) = reference(store, access, trace_id, trace_ref).await? else {
return Ok(None);
};
Ok(self
.current(store, access, trace_id, &trace_ref)
.await?
.map(|snapshot| snapshot.trace().clone()))
}
pub async fn get_trace_page<S: TraceStore>(
&self,
store: &S,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
cursor: Option<&str>,
page_size: u32,
) -> Result<Option<Trace>, ReadError<S::Error>> {
if !(1..=500).contains(&page_size) {
return Err(ReadError::InvalidParameters);
}
let Some(trace_ref) = reference(store, access, trace_id, trace_ref).await? else {
return Ok(None);
};
let Some(cursor) = cursor else {
let Some(snapshot) = self.current(store, access, trace_id, &trace_ref).await? else {
return Ok(None);
};
let position = SpanPosition {
trace_ref,
snapshot_ms: snapshot.snapshot_ms(),
offset: 0,
version: snapshot.version().to_owned(),
};
return page(&snapshot, &position, page_size, self.response_bytes).map(Some);
};
let position: SpanPosition = decode_cursor(cursor, "span")?;
if position.trace_ref != trace_ref || position.snapshot_ms == 0 {
return Err(ReadError::InvalidCursor("span"));
}
let Some(snapshot) = settle(
self.pinned(store, access, trace_id, &trace_ref, position.snapshot_ms)
.await,
)?
else {
return Ok(None);
};
if position.version != snapshot.version() {
return Err(ReadError::TraceChanged);
}
if position.offset > snapshot.trace().spans.len() {
return Err(ReadError::InvalidCursor("span"));
}
page(&snapshot, &position, page_size, self.response_bytes).map(Some)
}
pub(super) async fn current<S: TraceStore>(
&self,
store: &S,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
) -> Result<Option<Arc<Snapshot>>, ReadError<S::Error>> {
let latest = SnapshotKey::latest(store.source(), access, trace_id, trace_ref)?;
settle(
self.snapshots
.latest_or_load(latest, now_ms(), |snapshot_ms| {
self.pinned(store, access, trace_id, trace_ref, snapshot_ms)
})
.await,
)
}
async fn pinned<S: TraceStore>(
&self,
store: &S,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
snapshot_ms: u64,
) -> Result<Arc<Snapshot>, Arc<Miss<S::Error>>> {
let key = SnapshotKey::new(store.source(), access, trace_id, trace_ref, snapshot_ms)
.map_err(|error| Arc::new(error.into()))?;
self.snapshots
.pinned_or_load(key, snapshot_ms, async {
let params = TraceSpansParams {
access: access.clone(),
trace_id: trace_id.to_owned(),
trace_ref: trace_ref.to_owned(),
};
let rows = store
.trace_spans(&params, snapshot_ms)
.await
.map_err(|error| Miss::Read(map_store_error(error)))?;
let spend_rows = spend(store, access, &rows).await;
let freshness = Freshness::of(&rows, spend_rows.is_some(), snapshot_ms);
resolve_trace(
trace_id,
trace_ref,
&rows,
spend_rows.as_deref().unwrap_or_default(),
)
.map(|trace| (trace, freshness))
.ok_or(Miss::Absent)
})
.await
}
pub async fn get_span<S: TraceStore>(
&self,
store: &S,
access: &ReadAccessParams,
trace_id: &str,
span_id: &str,
trace_ref: &str,
) -> Result<Option<SpanDetail>, ReadError<S::Error>> {
let Some(trace_ref) = reference(store, access, trace_id, trace_ref).await? else {
return Ok(None);
};
let params = SpanDetailParams {
access: access.clone(),
trace_id: trace_id.to_owned(),
trace_ref,
span_id: span_id.to_owned(),
};
let row = store.span_detail(&params).await.map_err(map_store_error)?;
Ok(row.map(|row| SpanDetail {
input_ui: to_ui_content(&row.input),
output_ui: to_ui_content(&row.output),
span_id: row.span_id,
input: row.input,
output: row.output,
attributes: row.attributes,
}))
}
pub async fn get_span_error<S: TraceStore>(
&self,
store: &S,
access: &ReadAccessParams,
trace_id: &str,
span_id: &str,
trace_ref: &str,
cursor: Option<&str>,
) -> Result<Option<SpanErrorPage>, ReadError<S::Error>> {
let position = error_position(cursor)?;
let Some(trace_ref) = reference(store, access, trace_id, trace_ref).await? else {
return Ok(None);
};
let offset = position.as_ref().map_or(0, |position| position.offset);
let params = SpanErrorParams {
access: access.clone(),
trace_id: trace_id.to_owned(),
trace_ref,
span_id: span_id.to_owned(),
error_offset: offset,
error_version: position
.map(|position| position.version)
.unwrap_or_default(),
};
let Some(row) = store.span_error(&params).await.map_err(map_store_error)? else {
return Ok(None);
};
let next_offset = offset + row.message.chars().count() as u64;
let next_cursor = (next_offset < row.total_chars).then(|| {
encode_cursor(&ErrorPosition {
offset: next_offset,
version: row.version,
})
});
Ok(Some(SpanErrorPage {
span_id: row.span_id,
message: row.message,
total_chars: row.total_chars,
next_cursor,
}))
}
}
async fn reference<S: TraceStore>(
store: &S,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
) -> Result<Option<String>, ReadError<S::Error>> {
if !trace_ref.is_empty() {
return Ok(Some(trace_ref.to_owned()));
}
let params = TraceIdentityParams {
access: access.clone(),
trace_id: trace_id.to_owned(),
};
let identities = store.trace_refs(&params).await.map_err(map_store_error)?;
if identities.len() > 1 {
return Err(ReadError::AmbiguousTrace);
}
Ok(identities.into_iter().next())
}
fn page<E>(
snapshot: &Snapshot,
position: &SpanPosition,
page_size: u32,
response_bytes: usize,
) -> Result<Trace, ReadError<E>> {
let spans = &snapshot.trace().spans;
let create_page = |count: usize| {
let end = position.offset.saturating_add(count).min(spans.len());
Trace {
summary: snapshot.trace().summary.clone(),
agents: snapshot.trace().agents.clone(),
spans: spans[position.offset..end].to_vec(),
next_cursor: (end < spans.len()).then(|| {
encode_cursor(&SpanPosition {
trace_ref: position.trace_ref.clone(),
snapshot_ms: position.snapshot_ms,
offset: end,
version: snapshot.version().to_owned(),
})
}),
}
};
let mut trace = create_page(page_size as usize);
loop {
if serde_json::to_vec(&trace)
.map_err(|error| ReadError::Encode(Arc::new(error)))?
.len()
<= response_bytes
{
return Ok(trace);
}
if trace.spans.len() <= 1 {
return Err(ReadError::TooLarge);
}
trace = create_page(trace.spans.len() / 2);
}
}
pub(super) fn now_ms() -> u64 {
(time::OffsetDateTime::now_utc().unix_timestamp_nanos() / 1_000_000) as u64
}
pub(super) fn map_store_error<E>(error: StoreError<E>) -> ReadError<E> {
match error {
StoreError::TooLarge => ReadError::TooLarge,
StoreError::Failed(error) => ReadError::Store(Arc::new(error)),
}
}

View file

@ -0,0 +1,67 @@
use std::ops::Range;
use crate::TraceStore;
use litellm_traces::{
SpendLookup,
query::named::{
ReadAccessParams, SpendByResponseIdsParams, SpendByResponseIdsRow, TraceSpansRow,
},
};
const NANOS_PER_MS: i64 = 1_000_000;
const SPEND_WINDOW_MS: i64 = 30 * 60 * 1000;
pub(super) fn spend_window(rows: &[TraceSpansRow]) -> Option<Range<i64>> {
let start_ns = rows.iter().map(|row| row.start_ns).min()?;
let end_ns = rows
.iter()
.map(|row| row.start_ns.saturating_add_unsigned(row.duration_ns))
.max()?;
Some(
start_ns.div_euclid(NANOS_PER_MS) - SPEND_WINDOW_MS
..end_ns.div_euclid(NANOS_PER_MS) + SPEND_WINDOW_MS,
)
}
pub(super) fn spend_within(
spend: &[SpendByResponseIdsRow],
window: Range<i64>,
) -> &[SpendByResponseIdsRow] {
let first = spend.partition_point(|row| row.start_ms < window.start);
let end = spend.partition_point(|row| row.start_ms < window.end);
&spend[first..end.max(first)]
}
/// Spend rows sorted by `start_ms`, or `None` when the lookup failed and spend is unknown.
pub(super) async fn spend<S: TraceStore>(
store: &S,
access: &ReadAccessParams,
rows: &[TraceSpansRow],
) -> Option<Vec<SpendByResponseIdsRow>> {
let lookup = SpendLookup::new(rows);
let Some(window) = spend_window(rows) else {
return Some(Vec::new());
};
if lookup.is_empty() {
return Some(Vec::new());
}
let params = SpendByResponseIdsParams {
access: access.clone(),
response_ids: lookup.response_ids,
request_ids: lookup.request_ids,
trace_ids: lookup.trace_ids,
start_ms: window.start,
end_ms: window.end,
};
match store.spend(&params).await {
Ok(rows) => {
let mut rows = rows;
rows.sort_by_key(|row| row.start_ms);
Some(rows)
}
Err(error) => {
tracing::warn!(%error, "trace spend lookup unavailable");
None
}
}
}

View file

@ -0,0 +1,62 @@
use std::future::Future;
use litellm_traces::query::named::{
ListTracesParams, ListTracesRow, SpanDetailParams, SpanDetailRow, SpanErrorParams,
SpanErrorRow, SpendByResponseIdsParams, SpendByResponseIdsRow, TraceIdentityParams,
TracePageSpansParams, TraceSpansParams, TraceSpansRow,
};
#[derive(Debug, thiserror::Error)]
pub enum StoreError<E> {
#[error("trace read exceeds the storage read budget")]
TooLarge,
#[error(transparent)]
Failed(E),
}
pub trait TraceStore: Sync {
type Error: std::error::Error + Send + Sync + 'static;
/// Identifies the backing storage for snapshot cache keys.
fn source(&self) -> &str;
fn trace_refs(
&self,
params: &TraceIdentityParams,
) -> impl Future<Output = Result<Vec<String>, StoreError<Self::Error>>> + Send;
/// Returns `TooLarge` when the response exceeds the storage limit so the reader can halve `limit`.
fn list_runs(
&self,
params: &ListTracesParams,
) -> impl Future<Output = Result<Vec<ListTracesRow>, StoreError<Self::Error>>> + Send;
/// Returns spans visible at `snapshot_ms`, sorted by `start_ns`, or `TooLarge` past `MAX_GRAPH_BYTES`/`MAX_GRAPH_SPANS`.
fn trace_spans(
&self,
params: &TraceSpansParams,
snapshot_ms: u64,
) -> impl Future<Output = Result<Vec<TraceSpansRow>, StoreError<Self::Error>>> + Send;
/// Returns spans visible at `snapshot_ms`, sorted by `start_ns`, or `TooLarge` past `MAX_GRAPH_BYTES`/`MAX_GRAPH_SPANS`.
fn run_spans(
&self,
params: &TracePageSpansParams,
snapshot_ms: u64,
) -> impl Future<Output = Result<Vec<TraceSpansRow>, StoreError<Self::Error>>> + Send;
fn spend(
&self,
params: &SpendByResponseIdsParams,
) -> impl Future<Output = Result<Vec<SpendByResponseIdsRow>, StoreError<Self::Error>>> + Send;
fn span_detail(
&self,
params: &SpanDetailParams,
) -> impl Future<Output = Result<Option<SpanDetailRow>, StoreError<Self::Error>>> + Send;
fn span_error(
&self,
params: &SpanErrorParams,
) -> impl Future<Output = Result<Option<SpanErrorRow>, StoreError<Self::Error>>> + Send;
}

View file

@ -0,0 +1,768 @@
use std::{
collections::{HashMap, HashSet},
sync::{
Mutex,
atomic::{AtomicUsize, Ordering},
},
time::Duration,
};
use litellm_traces::{
CallEvidenceKind, CallKey, ObservationType, SpanStatus,
query::named::{
ListTracesParams, ListTracesRow, ReadAccessParams, SpanDetailParams, SpanDetailRow,
SpanErrorParams, SpanErrorRow, SpendByResponseIdsParams, SpendByResponseIdsRow,
TraceIdentityParams, TracePageSpansParams, TraceSpansParams, TraceSpansRow,
},
};
use litellm_traces_cache::{LIVE_TTL, ReadError, StoreError, TraceReader, TraceStore};
use rstest::rstest;
const START_NS: i64 = 1_790_742_989_000_000_000;
#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)]
enum Operation {
TraceRefs,
ListRuns,
TraceSpans,
RunSpans,
Spend,
SpanDetail,
SpanError,
}
#[derive(Clone, Copy)]
enum Failure {
TooLarge,
Failed,
}
#[derive(Debug, thiserror::Error)]
#[error("fake trace store failed")]
struct FakeError;
#[derive(Default)]
struct State {
failures: HashMap<Operation, Failure>,
trace_refs: Vec<String>,
list_runs: Vec<ListTracesRow>,
trace_spans: HashMap<String, Vec<TraceSpansRow>>,
run_spans: Vec<TraceSpansRow>,
spend: Vec<SpendByResponseIdsRow>,
span_detail: Option<SpanDetailRow>,
span_error: Option<SpanErrorRow>,
list_runs_too_large_above: Option<u32>,
trace_too_large_refs: HashSet<String>,
spend_fails_above_response_ids: Option<usize>,
}
#[derive(Default)]
struct Calls {
trace_refs: AtomicUsize,
list_runs: AtomicUsize,
trace_spans: AtomicUsize,
run_spans: AtomicUsize,
spend: AtomicUsize,
span_detail: AtomicUsize,
span_error: AtomicUsize,
}
#[derive(Default)]
struct FakeStore {
state: Mutex<State>,
calls: Calls,
}
impl FakeStore {
fn with_spans(trace_ref: &str, spans: Vec<TraceSpansRow>) -> Self {
Self {
state: Mutex::new(State {
trace_spans: HashMap::from([(trace_ref.to_owned(), spans)]),
..State::default()
}),
calls: Calls::default(),
}
}
fn set_failure(&self, operation: Operation, failure: Failure) {
self.state
.lock()
.unwrap()
.failures
.insert(operation, failure);
}
fn set_trace_refs(&self, trace_refs: Vec<String>) {
self.state.lock().unwrap().trace_refs = trace_refs;
}
fn set_list_runs(&self, rows: Vec<ListTracesRow>) {
self.state.lock().unwrap().list_runs = rows;
}
fn set_list_runs_too_large_above(&self, limit: u32) {
self.state.lock().unwrap().list_runs_too_large_above = Some(limit);
}
fn set_run_spans(&self, rows: Vec<TraceSpansRow>) {
self.state.lock().unwrap().run_spans = rows;
}
fn set_trace_spans_too_large(&self, trace_ref: &str) {
self.state
.lock()
.unwrap()
.trace_too_large_refs
.insert(trace_ref.to_owned());
}
/// Fails `spend` only when the lookup covers more than `limit` response ids, so a batch
/// covering several runs fails while each run's own narrower lookup still succeeds.
fn set_spend_fails_above_response_ids(&self, limit: usize) {
self.state.lock().unwrap().spend_fails_above_response_ids = Some(limit);
}
fn calls(&self, operation: Operation) -> usize {
match operation {
Operation::TraceRefs => self.calls.trace_refs.load(Ordering::SeqCst),
Operation::ListRuns => self.calls.list_runs.load(Ordering::SeqCst),
Operation::TraceSpans => self.calls.trace_spans.load(Ordering::SeqCst),
Operation::RunSpans => self.calls.run_spans.load(Ordering::SeqCst),
Operation::Spend => self.calls.spend.load(Ordering::SeqCst),
Operation::SpanDetail => self.calls.span_detail.load(Ordering::SeqCst),
Operation::SpanError => self.calls.span_error.load(Ordering::SeqCst),
}
}
fn failure(state: &State, operation: Operation) -> Result<(), StoreError<FakeError>> {
match state.failures.get(&operation) {
Some(Failure::TooLarge) => Err(StoreError::TooLarge),
Some(Failure::Failed) => Err(StoreError::Failed(FakeError)),
None => Ok(()),
}
}
}
impl TraceStore for FakeStore {
type Error = FakeError;
fn source(&self) -> &str {
"fake"
}
async fn trace_refs(
&self,
_: &TraceIdentityParams,
) -> Result<Vec<String>, StoreError<Self::Error>> {
self.calls.trace_refs.fetch_add(1, Ordering::SeqCst);
let state = self.state.lock().unwrap();
Self::failure(&state, Operation::TraceRefs)?;
Ok(state.trace_refs.clone())
}
async fn list_runs(
&self,
params: &ListTracesParams,
) -> Result<Vec<ListTracesRow>, StoreError<Self::Error>> {
self.calls.list_runs.fetch_add(1, Ordering::SeqCst);
let state = self.state.lock().unwrap();
Self::failure(&state, Operation::ListRuns)?;
if state
.list_runs_too_large_above
.is_some_and(|limit| params.limit > limit)
{
return Err(StoreError::TooLarge);
}
Ok(state
.list_runs
.iter()
.take(params.limit as usize)
.cloned()
.collect())
}
async fn trace_spans(
&self,
params: &TraceSpansParams,
_: u64,
) -> Result<Vec<TraceSpansRow>, StoreError<Self::Error>> {
self.calls.trace_spans.fetch_add(1, Ordering::SeqCst);
tokio::task::yield_now().await;
let state = self.state.lock().unwrap();
Self::failure(&state, Operation::TraceSpans)?;
if state.trace_too_large_refs.contains(&params.trace_ref) {
return Err(StoreError::TooLarge);
}
Ok(state
.trace_spans
.get(&params.trace_ref)
.cloned()
.unwrap_or_default())
}
async fn run_spans(
&self,
_: &TracePageSpansParams,
_: u64,
) -> Result<Vec<TraceSpansRow>, StoreError<Self::Error>> {
self.calls.run_spans.fetch_add(1, Ordering::SeqCst);
tokio::task::yield_now().await;
let state = self.state.lock().unwrap();
Self::failure(&state, Operation::RunSpans)?;
Ok(state.run_spans.clone())
}
async fn spend(
&self,
params: &SpendByResponseIdsParams,
) -> Result<Vec<SpendByResponseIdsRow>, StoreError<Self::Error>> {
self.calls.spend.fetch_add(1, Ordering::SeqCst);
let state = self.state.lock().unwrap();
Self::failure(&state, Operation::Spend)?;
if state
.spend_fails_above_response_ids
.is_some_and(|limit| params.response_ids.len() > limit)
{
return Err(StoreError::Failed(FakeError));
}
Ok(state.spend.clone())
}
async fn span_detail(
&self,
_: &SpanDetailParams,
) -> Result<Option<SpanDetailRow>, StoreError<Self::Error>> {
self.calls.span_detail.fetch_add(1, Ordering::SeqCst);
let state = self.state.lock().unwrap();
Self::failure(&state, Operation::SpanDetail)?;
Ok(state.span_detail.clone())
}
async fn span_error(
&self,
_: &SpanErrorParams,
) -> Result<Option<SpanErrorRow>, StoreError<Self::Error>> {
self.calls.span_error.fetch_add(1, Ordering::SeqCst);
let state = self.state.lock().unwrap();
Self::failure(&state, Operation::SpanError)?;
Ok(state.span_error.clone())
}
}
fn access() -> ReadAccessParams {
ReadAccessParams {
all_teams: true,
user_id: String::new(),
team_ids: Vec::new(),
}
}
fn span(index: usize) -> TraceSpansRow {
TraceSpansRow {
trace_id: "trace".into(),
span_id: format!("span-{index}"),
parent_span_id: if index == 0 {
String::new()
} else {
"span-0".into()
},
name: "agent".into(),
kind: ObservationType::Agent,
wrapper_candidate: false,
agent: "agent".into(),
framework: String::new(),
status: SpanStatus::Ok,
status_message: String::new(),
error_truncated: false,
start_ns: START_NS + index as i64 * 1_000_000,
duration_ns: 10_000_000,
service: "test".into(),
input_preview: format!("span input {index}"),
model: String::new(),
input_tokens: 0,
output_tokens: 0,
litellm_request_id: String::new(),
call_keys: Vec::new(),
call_evidence: None,
tool_call_id: String::new(),
team_id: "team".into(),
api_key_hash: "key".into(),
user_id: "user".into(),
}
}
fn run(trace_id: &str, trace_ref: &str) -> ListTracesRow {
ListTracesRow {
trace_id: trace_id.into(),
trace_ref: trace_ref.into(),
team_id: "team".into(),
api_key_hash: "key".into(),
user_id: "user".into(),
name: "listed".into(),
service: "test".into(),
input_preview: String::new(),
status: SpanStatus::Ok,
start_ms: 1_790_742_989_000,
duration_ms: 10,
span_count: 1,
agent_count: 1,
agent_invocations: 1,
agent_names: vec!["agent".into()],
frameworks: Vec::new(),
llm_calls: 0,
tool_calls: 0,
input_tokens: 0,
output_tokens: 0,
models: Vec::new(),
error_count: 0,
request_ids: Vec::new(),
}
}
#[rstest]
#[tokio::test]
async fn pages_reuse_one_trace_snapshot_and_concatenate_in_order() {
let store = FakeStore::with_spans("ref", (0..5).map(span).collect());
let reader = TraceReader::new(usize::MAX);
let access = access();
let first = reader
.get_trace_page(&store, &access, "trace", "ref", None, 2)
.await
.unwrap()
.unwrap();
let second = reader
.get_trace_page(
&store,
&access,
"trace",
"ref",
first.next_cursor.as_deref(),
2,
)
.await
.unwrap()
.unwrap();
let third = reader
.get_trace_page(
&store,
&access,
"trace",
"ref",
second.next_cursor.as_deref(),
2,
)
.await
.unwrap()
.unwrap();
let ids: Vec<_> = first
.spans
.iter()
.chain(&second.spans)
.chain(&third.spans)
.map(|span| span.span_id.as_str())
.collect();
assert_eq!(ids, ["span-0", "span-1", "span-2", "span-3", "span-4"]);
assert!(third.next_cursor.is_none());
assert_eq!(store.calls(Operation::TraceSpans), 1);
}
#[rstest]
#[tokio::test]
async fn snapshot_versions_are_stable_across_readers_and_detect_changes() {
let access = access();
let original = FakeStore::with_spans("ref", vec![span(0), span(1)]);
let reader_a = TraceReader::new(usize::MAX);
let first = reader_a
.get_trace_page(&original, &access, "trace", "ref", None, 1)
.await
.unwrap()
.unwrap();
let cursor = first.next_cursor.unwrap();
let changed = FakeStore::with_spans("ref", vec![span(0), span(1), span(2)]);
let reader_b = TraceReader::new(usize::MAX);
let result = reader_b
.get_trace_page(&changed, &access, "trace", "ref", Some(&cursor), 1)
.await;
assert!(matches!(result, Err(ReadError::TraceChanged)));
let unchanged = FakeStore::with_spans("ref", vec![span(0), span(1)]);
let reader_c = TraceReader::new(usize::MAX);
let next = reader_c
.get_trace_page(&unchanged, &access, "trace", "ref", Some(&cursor), 1)
.await
.unwrap()
.unwrap();
assert_eq!(next.spans[0].span_id, "span-1");
}
#[rstest]
#[tokio::test]
async fn response_size_splits_pages_and_rejects_a_single_oversized_span() {
let spans: Vec<_> = (0..4)
.map(|index| {
let mut row = span(index);
row.input_preview = "x".repeat(256);
row
})
.collect();
let access = access();
let full_budget_reader = TraceReader::new(usize::MAX);
let one_span = full_budget_reader
.get_trace_page(
&FakeStore::with_spans("ref", spans.clone()),
&access,
"trace",
"ref",
None,
1,
)
.await
.unwrap()
.unwrap();
let response_bytes = serde_json::to_vec(&one_span).unwrap().len() + 128;
let reader = TraceReader::new(response_bytes);
let store = FakeStore::with_spans("ref", spans.clone());
let page = reader
.get_trace_page(&store, &access, "trace", "ref", None, 4)
.await
.unwrap()
.unwrap();
assert!(!page.spans.is_empty());
assert!(page.spans.len() < 4);
assert!(page.next_cursor.is_some());
let continued = reader
.get_trace_page(
&store,
&access,
"trace",
"ref",
page.next_cursor.as_deref(),
4,
)
.await
.unwrap()
.unwrap();
assert!(!continued.spans.is_empty());
assert!(matches!(
TraceReader::new(1)
.get_trace_page(&store, &access, "trace", "ref", None, 1)
.await,
Err(ReadError::TooLarge)
));
}
#[rstest]
#[tokio::test]
async fn list_run_budget_halves_the_limit_and_cursor_requires_a_full_page() {
let store = FakeStore::default();
store.set_list_runs(
(0..3)
.map(|index| run(&format!("trace-{index}"), &format!("ref-{index}")))
.collect(),
);
store.set_list_runs_too_large_above(2);
let reader = TraceReader::new(usize::MAX);
let access = access();
let page = reader
.list_traces(&store, &access, 0, i64::MAX, None, 8)
.await
.unwrap();
assert_eq!(page.data.len(), 2);
assert!(page.next_cursor.is_some());
assert_eq!(store.calls(Operation::ListRuns), 3);
let shorter = FakeStore::default();
shorter.set_list_runs(vec![run("only", "ref-only")]);
shorter.set_list_runs_too_large_above(2);
let page = reader
.list_traces(&shorter, &access, 0, i64::MAX, None, 8)
.await
.unwrap();
assert_eq!(page.data.len(), 1);
assert!(page.next_cursor.is_none());
assert_eq!(shorter.calls(Operation::ListRuns), 1);
}
#[rstest]
#[tokio::test]
async fn oversized_run_batch_falls_back_to_each_run_and_keeps_listed_summaries() {
let store = FakeStore::with_spans("ref-good", vec![span(0)]);
store.set_list_runs(vec![
run("trace-large", "ref-large"),
run("trace-good", "ref-good"),
]);
store.set_trace_spans_too_large("ref-large");
store.set_run_spans(Vec::new());
store.set_failure(Operation::RunSpans, Failure::TooLarge);
let reader = TraceReader::new(usize::MAX);
let page = reader
.list_traces(&store, &access(), 0, i64::MAX, None, 2)
.await
.unwrap();
assert_eq!(page.data.len(), 2);
assert!(page.data[0].resolution_limited);
assert_eq!(page.data[0].trace_ref, "ref-large");
assert!(!page.data[1].resolution_limited);
assert_eq!(page.data[1].trace_ref, "ref-good");
assert_eq!(store.calls(Operation::RunSpans), 1);
assert_eq!(store.calls(Operation::TraceSpans), 2);
let again = reader
.list_traces(&store, &access(), 0, i64::MAX, None, 2)
.await
.unwrap();
assert_eq!(again.data, page.data);
assert_eq!(store.calls(Operation::RunSpans), 1);
assert_eq!(store.calls(Operation::TraceSpans), 2);
}
fn spend_row(response_id: &str, cost: f64) -> SpendByResponseIdsRow {
SpendByResponseIdsRow {
request_id: response_id.into(),
litellm_call_id: String::new(),
response_id: response_id.into(),
upstream_response_id: String::new(),
trace_id: String::new(),
span_id: String::new(),
team_id: "team".into(),
api_key: "key".into(),
user: "user".into(),
spend: Some(cost),
start_ms: START_NS / 1_000_000,
}
}
#[rstest]
#[tokio::test]
async fn failed_batch_spend_lookup_falls_back_to_each_run_instead_of_losing_every_cost() {
let mut first = span(0);
first.trace_id = "trace-a".into();
first.kind = ObservationType::Llm;
first.litellm_request_id = "response-a".into();
first.call_keys = vec![CallKey::ProviderResponse("response-a".into())];
first.call_evidence = Some(CallEvidenceKind::Complete);
let mut second = span(0);
second.trace_id = "trace-b".into();
second.kind = ObservationType::Llm;
second.litellm_request_id = "response-b".into();
second.call_keys = vec![CallKey::ProviderResponse("response-b".into())];
second.call_evidence = Some(CallEvidenceKind::Complete);
let store = FakeStore::default();
store.set_list_runs(vec![run("trace-a", "ref-a"), run("trace-b", "ref-b")]);
store.set_run_spans(vec![first.clone(), second.clone()]);
{
let mut state = store.state.lock().unwrap();
state.trace_spans.insert("ref-a".to_owned(), vec![first]);
state.trace_spans.insert("ref-b".to_owned(), vec![second]);
state.spend = vec![spend_row("response-a", 1.5), spend_row("response-b", 2.5)];
}
// The batch covers both runs' response ids (2); each run resolved on its own only ever
// asks for its own (1), so this fails only the combined read, not the per-run fallback.
store.set_spend_fails_above_response_ids(1);
let page = TraceReader::new(usize::MAX)
.list_traces(&store, &access(), 0, i64::MAX, None, 8)
.await
.unwrap();
assert_eq!(page.data.len(), 2);
let by_ref: HashMap<&str, f64> = page
.data
.iter()
.map(|run| {
(
run.trace_ref.as_str(),
run.spend
.expect("run's own spend read should have succeeded"),
)
})
.collect();
assert_eq!(by_ref["ref-a"], 1.5);
assert_eq!(by_ref["ref-b"], 2.5);
assert_eq!(store.calls(Operation::RunSpans), 1);
assert_eq!(store.calls(Operation::TraceSpans), 2);
}
#[rstest]
#[tokio::test]
async fn failed_spend_lookup_preserves_the_trace_with_unknown_spend() {
let mut row = span(0);
row.litellm_request_id = "response".into();
row.call_keys = vec![CallKey::ProviderResponse("response".into())];
row.call_evidence = Some(CallEvidenceKind::Complete);
let store = FakeStore::with_spans("ref", vec![row]);
store.set_failure(Operation::Spend, Failure::Failed);
let trace = TraceReader::new(usize::MAX)
.get_trace(&store, &access(), "trace", "ref")
.await
.unwrap()
.unwrap();
assert_eq!(trace.summary.spend, None);
assert_eq!(trace.spans[0].spend, None);
assert_eq!(store.calls(Operation::Spend), 1);
}
#[rstest]
#[tokio::test]
async fn ambiguous_trace_references_fail_and_a_single_reference_is_resolved() {
let reader = TraceReader::new(usize::MAX);
let access = access();
let ambiguous = FakeStore::default();
ambiguous.set_trace_refs(vec!["ref-a".into(), "ref-b".into()]);
assert!(matches!(
reader.get_trace(&ambiguous, &access, "trace", "").await,
Err(ReadError::AmbiguousTrace)
));
let unique = FakeStore::with_spans("ref-only", vec![span(0)]);
unique.set_trace_refs(vec!["ref-only".into()]);
let trace = reader
.get_trace(&unique, &access, "trace", "")
.await
.unwrap()
.unwrap();
assert_eq!(trace.summary.trace_ref, "ref-only");
}
#[rstest]
#[case::zero(0)]
#[case::above_max(501)]
#[tokio::test]
async fn invalid_page_sizes_are_rejected(#[case] page_size: u32) {
let store = FakeStore::default();
let reader = TraceReader::new(usize::MAX);
let access = access();
assert!(matches!(
reader
.get_trace_page(&store, &access, "trace", "ref", None, page_size)
.await,
Err(ReadError::InvalidParameters)
));
}
#[rstest]
#[tokio::test]
async fn zero_list_limit_is_rejected() {
let store = FakeStore::default();
let reader = TraceReader::new(usize::MAX);
let access = access();
assert!(matches!(
reader
.list_traces(&store, &access, 0, i64::MAX, None, 0)
.await,
Err(ReadError::InvalidParameters)
));
}
fn now_ns() -> i64 {
time::OffsetDateTime::now_utc().unix_timestamp_nanos() as i64
}
#[rstest]
#[tokio::test]
async fn concurrent_and_repeated_opens_share_one_storage_read() {
let store = FakeStore::with_spans("ref", (0..3).map(span).collect());
let reader = TraceReader::new(usize::MAX);
let access = access();
let (first, second) = tokio::join!(
reader.get_trace(&store, &access, "trace", "ref"),
reader.get_trace_page(&store, &access, "trace", "ref", None, 2),
);
let first = first.unwrap().unwrap();
let second = second.unwrap().unwrap();
let reopened = reader
.get_trace_page(&store, &access, "trace", "ref", None, 2)
.await
.unwrap()
.unwrap();
assert_eq!(first.spans.len(), 3);
assert_eq!(second.spans, first.spans[..2]);
assert_eq!(reopened.next_cursor, second.next_cursor);
assert_eq!(store.calls(Operation::TraceSpans), 1);
}
#[rstest]
#[tokio::test]
async fn failed_reads_are_not_cached() {
let store = FakeStore::with_spans("ref", vec![span(0)]);
store.set_failure(Operation::TraceSpans, Failure::Failed);
let reader = TraceReader::new(usize::MAX);
let access = access();
assert!(matches!(
reader.get_trace(&store, &access, "trace", "ref").await,
Err(ReadError::Store(_))
));
store.state.lock().unwrap().failures.clear();
let trace = reader
.get_trace(&store, &access, "trace", "ref")
.await
.unwrap()
.unwrap();
assert_eq!(trace.spans.len(), 1);
assert_eq!(store.calls(Operation::TraceSpans), 2);
}
#[rstest]
#[tokio::test]
async fn listed_runs_are_read_once_until_a_live_run_expires() {
let live = TraceSpansRow {
trace_id: "trace-live".into(),
start_ns: now_ns(),
..span(0)
};
let settled = TraceSpansRow {
trace_id: "trace-settled".into(),
..span(0)
};
let store = FakeStore::default();
store.set_list_runs(vec![
run("trace-live", "ref-live"),
run("trace-settled", "ref-settled"),
]);
store.set_run_spans(vec![live, settled]);
let reader = TraceReader::new(usize::MAX);
let access = access();
let list = || reader.list_traces(&store, &access, 0, i64::MAX, None, 2);
let first = list().await.unwrap();
assert!(first.data.iter().all(|summary| summary.name == "agent"));
list().await.unwrap();
assert_eq!(store.calls(Operation::RunSpans), 1);
store.set_run_spans(Vec::new());
tokio::time::sleep(LIVE_TTL + Duration::from_millis(200)).await;
let after = list().await.unwrap();
assert_eq!(store.calls(Operation::RunSpans), 2);
assert_eq!(after.data[0].name, "listed");
assert_eq!(after.data[1], first.data[1]);
}
#[rstest]
#[tokio::test]
async fn concurrent_pages_of_an_evicted_snapshot_share_one_storage_read() {
let access = access();
let first = TraceReader::new(usize::MAX)
.get_trace_page(
&FakeStore::with_spans("ref", (0..3).map(span).collect()),
&access,
"trace",
"ref",
None,
1,
)
.await
.unwrap()
.unwrap();
let cursor = first.next_cursor.as_deref();
let store = FakeStore::with_spans("ref", (0..3).map(span).collect());
let reader = TraceReader::new(usize::MAX);
let (left, right) = tokio::join!(
reader.get_trace_page(&store, &access, "trace", "ref", cursor, 1),
reader.get_trace_page(&store, &access, "trace", "ref", cursor, 1),
);
assert_eq!(left.unwrap().unwrap().spans[0].span_id, "span-1");
assert_eq!(right.unwrap().unwrap().spans[0].span_id, "span-1");
assert_eq!(store.calls(Operation::TraceSpans), 1);
}

View file

@ -0,0 +1,211 @@
use std::time::Duration;
use litellm_traces::{
SpanStatus, Trace,
query::named::{ReadAccessParams, SpendByResponseIdsRow, TraceSpansRow},
resolve_trace,
};
use std::sync::Arc;
use litellm_traces_cache::{Error, Freshness, Snapshot, SnapshotCache, SnapshotKey};
use rstest::{fixture, rstest};
const T0: i64 = 1_790_742_989_000_000_000;
const MS: i64 = 1_000_000;
const TTL: Duration = Duration::from_secs(120);
fn row(span_id: &str, parent: &str, name: &str, kind: &str, agent: &str) -> TraceSpansRow {
TraceSpansRow {
trace_id: String::new(),
span_id: span_id.into(),
parent_span_id: parent.into(),
name: name.into(),
kind: kind.parse().unwrap(),
wrapper_candidate: false,
agent: agent.into(),
framework: String::new(),
status: SpanStatus::Ok,
status_message: String::new(),
error_truncated: false,
start_ns: T0,
duration_ns: 10 * MS as u64,
service: "agent-demo".into(),
input_preview: format!("input of {name}"),
model: String::new(),
input_tokens: 0,
output_tokens: 0,
litellm_request_id: String::new(),
call_keys: Vec::new(),
call_evidence: None,
tool_call_id: String::new(),
team_id: String::new(),
api_key_hash: String::new(),
user_id: String::new(),
}
}
fn access() -> ReadAccessParams {
ReadAccessParams {
all_teams: false,
user_id: String::new(),
team_ids: vec!["team".into()],
}
}
fn key(
source: &str,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
ms: u64,
) -> SnapshotKey {
SnapshotKey::new(source, access, trace_id, trace_ref, ms).unwrap()
}
async fn insert(
cache: &SnapshotCache,
key: SnapshotKey,
trace: Trace,
) -> Result<Arc<Snapshot>, Arc<Error>> {
cache
.pinned_or_load(key, 100, async {
Ok::<_, Error>((trace, Freshness::Settled))
})
.await
}
#[fixture]
fn trace() -> Trace {
resolve_trace(
"trace",
"ref",
&[row("root", "", "run", "agent", "agent")],
&[] as &[SpendByResponseIdsRow],
)
.expect("fixture should resolve")
}
#[rstest]
#[case::different_team(false, "", "other-team")]
#[case::different_user(false, "other-user", "team")]
#[case::different_scope(true, "", "team")]
#[tokio::test]
async fn cached_trace_is_isolated_by_access_scope(
trace: Trace,
#[case] all_teams: bool,
#[case] user_id: &str,
#[case] team_id: &str,
) {
let cache = SnapshotCache::new(1024 * 1024, TTL);
let stored = key("source", &access(), "trace", "ref", 100);
insert(&cache, stored.clone(), trace.clone()).await.unwrap();
let other_access = ReadAccessParams {
all_teams,
user_id: user_id.into(),
team_ids: vec![team_id.into()],
};
let other = key("source", &other_access, "trace", "ref", 100);
assert!(cache.get(&other).await.is_none());
let cached = cache.get(&stored).await.unwrap();
assert_eq!(cached.trace(), &trace);
}
#[rstest]
#[case::different_source("other-source", "trace", "ref", 100)]
#[case::different_trace_id("source", "other-trace", "ref", 100)]
#[case::different_trace_ref("source", "trace", "other-ref", 100)]
#[case::different_snapshot_ms("source", "trace", "ref", 200)]
#[tokio::test]
async fn cached_trace_is_isolated_by_key_fields(
trace: Trace,
#[case] source: &str,
#[case] trace_id: &str,
#[case] trace_ref: &str,
#[case] snapshot_ms: u64,
) {
let cache = SnapshotCache::new(1024 * 1024, TTL);
let stored = key("source", &access(), "trace", "ref", 100);
insert(&cache, stored.clone(), trace.clone()).await.unwrap();
let other = key(source, &access(), trace_id, trace_ref, snapshot_ms);
assert!(cache.get(&other).await.is_none());
assert!(cache.get(&stored).await.is_some());
}
#[rstest]
#[tokio::test]
async fn snapshot_at_the_size_limit_is_accepted(trace: Trace) {
let size = serde_json::to_vec(&trace).unwrap().len();
let cache = SnapshotCache::new(size, TTL);
let stored = key("source", &access(), "trace", "ref", 100);
insert(&cache, stored.clone(), trace).await.unwrap();
assert!(cache.get(&stored).await.is_some());
}
#[rstest]
#[tokio::test]
async fn snapshot_one_byte_over_the_size_limit_is_rejected(trace: Trace) {
let size = serde_json::to_vec(&trace).unwrap().len();
let cache = SnapshotCache::new(size - 1, TTL);
let stored = key("source", &access(), "trace", "ref", 100);
assert!(matches!(
insert(&cache, stored.clone(), trace).await,
Err(error) if matches!(*error, Error::ReadTooLarge)
));
assert!(cache.get(&stored).await.is_none());
}
#[rstest]
#[case::same_ids(&["root", "child"], &["root", "child"], true)]
#[case::different_ids(&["root", "child"], &["root", "other"], false)]
#[tokio::test]
async fn snapshot_version_tracks_the_ordered_span_ids(
#[case] first_ids: &[&str],
#[case] second_ids: &[&str],
#[case] equal: bool,
) {
let build = |ids: &[&str]| -> Trace {
let rows: Vec<TraceSpansRow> = ids
.iter()
.map(|span_id| row(span_id, "", "run", "agent", "agent"))
.collect();
resolve_trace("trace", "ref", &rows, &[] as &[SpendByResponseIdsRow])
.expect("fixture should resolve")
};
let cache = SnapshotCache::new(1024 * 1024, TTL);
let first = insert(
&cache,
key("source", &access(), "a", "ref", 100),
build(first_ids),
)
.await
.unwrap();
let second = insert(
&cache,
key("source", &access(), "b", "ref", 100),
build(second_ids),
)
.await
.unwrap();
assert_eq!(first.version() == second.version(), equal);
}
#[rstest]
#[tokio::test]
async fn snapshots_expire_when_idle(trace: Trace) {
let cache = SnapshotCache::new(1024 * 1024, Duration::from_millis(50));
let stored = key("source", &access(), "trace", "ref", 100);
insert(&cache, stored.clone(), trace).await.unwrap();
tokio::time::sleep(Duration::from_millis(200)).await;
assert!(cache.get(&stored).await.is_none());
}

View file

@ -1,6 +1,7 @@
- Own trace schema, row encoding, SQL query adapters and reader provisioning; consume domain types from `litellm-traces`
- Keep generic ClickHouse connections and HTTP execution in `litellm-storage-clickhouse`; keep PyO3 conversion in `python-bridge`
- Keep schema definitions only in `migrations/NNNN_description.sql`, embedded by `litellm_migrate::migrate!`
- Keep schema definitions only in `migrations/NNNN_description.sql`, embedded by `sqlx::migrate!`
- Treat retention TTLs as current configuration: change them in `RETENTION` in `src/schema.rs`, which every startup reapplies, never in a new migration
- Require typed query parameters and SELECT-only readers with server-side limits and tenant isolation
- Bound insert time and encoded bytes; preserve shared values and explicit retry deduplication
- Test storage behavior through the public API against ClickHouse

View file

@ -12,23 +12,22 @@ schema = ["dep:schemars", "litellm-traces/schema"]
macro_rules_attribute.workspace = true
schemars = { workspace = true, optional = true }
askama.workspace = true
base64.workspace = true
flate2.workspace = true
futures-util.workspace = true
hmac = "0.12.1"
litellm-http.workspace = true
litellm-migrate.workspace = true
litellm-storage-clickhouse.workspace = true
litellm-traces.workspace = true
litellm-traces-cache.workspace = true
moka.workspace = true
serde.workspace = true
serde_json.workspace = true
sha2.workspace = true
sqlx = { workspace = true, features = ["migrate", "macros"] }
strum.workspace = true
thiserror.workspace = true
time = { workspace = true, features = ["formatting"] }
tokio.workspace = true
tracing.workspace = true
url.workspace = true
[dev-dependencies]

View file

@ -0,0 +1,6 @@
ALTER TABLE {database}.spend_logs
ADD COLUMN IF NOT EXISTS litellm_call_id String DEFAULT '' AFTER response_id,
ADD INDEX IF NOT EXISTS idx_litellm_call_id litellm_call_id
TYPE bloom_filter(0.001) GRANULARITY 1,
ADD INDEX IF NOT EXISTS idx_request_id request_id
TYPE bloom_filter(0.001) GRANULARITY 1

View file

@ -0,0 +1,14 @@
SELECT t.TraceId, t.SpanId, s.request_id, s.spend, s.metadata
FROM otel_traces AS t
INNER JOIN (
SELECT *
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
) AS s
ON t.LiteLLMRequestId = s.response_id
AND t.TeamId = s.team_id
AND ((t.UserId != '' AND t.UserId = s.user)
OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key))
WHERE t.Timestamp >= now() - INTERVAL 1 DAY
AND t.LiteLLMRequestId != ''
LIMIT 100

View file

@ -0,0 +1,8 @@
SELECT
request_id, response_id, model, spend, JSONExtractString(metadata, 'project') AS project
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
AND JSONHas(metadata, 'project')
AND JSONExtractString(metadata, 'project') = 'example'
ORDER BY start_time DESC
LIMIT 100

View file

@ -0,0 +1,6 @@
SELECT
DISTINCT arrayJoin(JSONExtractKeys(metadata)) AS key
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 30 DAY
ORDER BY key
LIMIT 200

View file

@ -0,0 +1,7 @@
SELECT TeamId AS team, ApiKeyHash AS api_key, TraceId AS trace_id,
SpanId AS span_id, StatusMessage AS message
FROM otel_traces
WHERE Timestamp >= now() - INTERVAL 1 DAY
AND StatusCode = 'STATUS_CODE_ERROR'
ORDER BY Timestamp DESC, team, api_key, trace_id, span_id
LIMIT 100

View file

@ -1,5 +1,8 @@
SELECT team_id AS team, api_key, request_id, spend,
JSONExtractString(metadata, 'labels', 'priority') AS priority
FROM spend_logs FINAL
WHERE JSONExtractString(metadata, 'labels', 'priority') = 'high'
WHERE start_time >= now() - INTERVAL 1 DAY
AND JSONHas(metadata, 'labels', 'priority')
AND JSONExtractString(metadata, 'labels', 'priority') = 'high'
ORDER BY team, api_key, request_id
LIMIT 100

View file

@ -0,0 +1,17 @@
SELECT
team_id, model, requests, unknown_cost_requests,
if(unknown_cost_requests = 0, recorded_spend, NULL) AS spend,
input_tokens, output_tokens
FROM (
SELECT
team_id, model, count() AS requests,
countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests,
sum(spend) AS recorded_spend,
sum(prompt_tokens) AS input_tokens,
sum(completion_tokens) AS output_tokens
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
GROUP BY team_id, model
)
ORDER BY team_id, model
LIMIT 100

View file

@ -0,0 +1,8 @@
SELECT
request_id,
JSONType(metadata, 'labels', 'priority') AS type,
JSONExtractRaw(metadata, 'labels', 'priority') AS value
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
AND JSONHas(metadata, 'labels', 'priority')
LIMIT 100

View file

@ -0,0 +1,8 @@
SELECT
TraceId, SpanId, Model, InputTokens, OutputTokens,
Duration / 1000000 AS duration_ms
FROM otel_traces
WHERE Timestamp >= now() - INTERVAL 1 DAY
AND ObservationType = 'llm'
ORDER BY Timestamp DESC
LIMIT 100

View file

@ -0,0 +1,7 @@
SELECT
request_id, response_id, trace_id, span_id, model, spend,
prompt_tokens, completion_tokens, status
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
ORDER BY start_time DESC, request_id
LIMIT 100

Some files were not shown because too many files have changed in this diff Show more