mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Merge origin/litellm_internal_staging into litellm_lit4395_cursor_agent
This commit is contained in:
commit
9eeff06263
1122 changed files with 21336 additions and 6074 deletions
|
|
@ -88,6 +88,36 @@ commands:
|
|||
rm -f /tmp/uv-install.sh
|
||||
echo 'export PATH="$HOME/.local/bin:$PATH"' >> "$BASH_ENV"
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
install_rust:
|
||||
description: "Install pinned rustup (1.28.2) and Rust toolchain (1.97.1) with checksum verification. Adds ~/.cargo/bin to PATH. Run this before any `uv sync` or `uv build` of the workspace: the root package builds litellm-rust through maturin, and on an image without cargo maturin fetches an unpinned rustup and a floating toolchain by itself."
|
||||
steps:
|
||||
- run:
|
||||
name: Install Rust (rustup 1.28.2, toolchain 1.97.1)
|
||||
command: |
|
||||
case "$(uname -m)" in
|
||||
x86_64)
|
||||
RUSTUP_TRIPLE=x86_64-unknown-linux-gnu
|
||||
RUSTUP_SHA256=20a06e644b0d9bd2fbdbfd52d42540bdde820ea7df86e92e533c073da0cdd43c
|
||||
;;
|
||||
aarch64)
|
||||
RUSTUP_TRIPLE=aarch64-unknown-linux-gnu
|
||||
RUSTUP_SHA256=e3853c5a252fca15252d07cb23a1bdd9377a8c6f3efa01531109281ae47f841c
|
||||
;;
|
||||
*)
|
||||
echo "install_rust: unsupported architecture $(uname -m)" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
curl -sSLf -o /tmp/rustup-init \
|
||||
"https://static.rust-lang.org/rustup/archive/1.28.2/${RUSTUP_TRIPLE}/rustup-init"
|
||||
echo "${RUSTUP_SHA256} /tmp/rustup-init" | sha256sum -c -
|
||||
chmod +x /tmp/rustup-init
|
||||
/tmp/rustup-init -y --no-modify-path --profile minimal --default-toolchain 1.97.1
|
||||
rm -f /tmp/rustup-init
|
||||
echo 'export PATH="$HOME/.cargo/bin:$PATH"' >> "$BASH_ENV"
|
||||
export PATH="$HOME/.cargo/bin:$PATH"
|
||||
rustc --version
|
||||
cargo --version
|
||||
start_postgres:
|
||||
description: "Start a postgres-db container on port 5432 and wait until it accepts connections."
|
||||
parameters:
|
||||
|
|
@ -163,6 +193,26 @@ commands:
|
|||
done
|
||||
echo "fake OpenAI endpoint did not become ready" >&2
|
||||
exit 1
|
||||
start_cost_center_service:
|
||||
description: "Start the stand-in cost center validation service (tests/store_model_in_db_tests/cost_center_service.py) on host port 9414 and wait until healthy. The proxy's team-metadata validator (team_metadata_validator_e2e.py, impl 'http') reaches it via TEAM_METADATA_VALIDATION_SERVICE_URL=http://host.docker.internal:9414/validate. Run after uv deps are synced."
|
||||
steps:
|
||||
- run:
|
||||
name: Start cost center validation service
|
||||
background: true
|
||||
command: |
|
||||
uv run --no-sync python tests/store_model_in_db_tests/cost_center_service.py --host 0.0.0.0 --port 9414
|
||||
- run:
|
||||
name: Wait for cost center validation service
|
||||
command: |
|
||||
for i in $(seq 1 30); do
|
||||
if curl -sf http://localhost:9414/health >/dev/null 2>&1; then
|
||||
echo "cost center validation service is up"
|
||||
exit 0
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
echo "cost center validation service did not become ready" >&2
|
||||
exit 1
|
||||
setup_litellm_enterprise_pip:
|
||||
steps:
|
||||
- run:
|
||||
|
|
@ -178,6 +228,7 @@ commands:
|
|||
- checkout
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
|
|
@ -292,6 +343,7 @@ jobs:
|
|||
- checkout
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Build the wheel
|
||||
environment:
|
||||
|
|
@ -324,6 +376,7 @@ jobs:
|
|||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -397,6 +450,7 @@ jobs:
|
|||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -471,6 +525,7 @@ jobs:
|
|||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -522,6 +577,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -588,6 +644,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -628,6 +685,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -669,6 +727,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -702,6 +761,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
|
|
@ -752,6 +812,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
|
|
@ -803,6 +864,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -836,6 +898,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
|
|
@ -882,6 +945,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -928,6 +992,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -970,6 +1035,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1016,6 +1082,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1063,6 +1130,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
|
|
@ -1103,6 +1171,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1148,6 +1217,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1192,6 +1262,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1224,6 +1295,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1267,6 +1339,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1311,6 +1384,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1355,6 +1429,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1386,6 +1461,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1432,6 +1508,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1477,6 +1554,7 @@ jobs:
|
|||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1527,6 +1605,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1551,6 +1630,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1577,6 +1657,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1678,6 +1759,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1773,6 +1855,7 @@ jobs:
|
|||
at: ~/project
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1861,6 +1944,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -1944,6 +2028,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -2076,6 +2161,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -2162,6 +2248,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -2258,12 +2345,14 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
uv sync --frozen --all-groups --all-extras --python 3.12
|
||||
- start_postgres
|
||||
- start_fake_openai_endpoint
|
||||
- start_cost_center_service
|
||||
- attach_workspace:
|
||||
at: ~/project
|
||||
- run:
|
||||
|
|
@ -2283,11 +2372,13 @@ jobs:
|
|||
-e STORE_MODEL_IN_DB="True" \
|
||||
-e LITELLM_MASTER_KEY="sk-1234" \
|
||||
-e FAKE_OPENAI_API_BASE=http://host.docker.internal:8190 \
|
||||
-e TEAM_METADATA_VALIDATION_SERVICE_URL=http://host.docker.internal:9414/validate \
|
||||
-e LITELLM_LICENSE=$LITELLM_LICENSE \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/store_model_db_config.yaml:/app/config.yaml \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/team_metadata_validator_e2e.py:/app/team_metadata_validator_e2e.py \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000
|
||||
|
|
@ -2333,6 +2424,7 @@ jobs:
|
|||
- setup_google_dns
|
||||
# Remove Docker CLI installation since it's already available in machine executor
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -2414,6 +2506,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -2553,6 +2646,7 @@ jobs:
|
|||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
|
|
@ -2743,6 +2837,7 @@ jobs:
|
|||
category: client
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
|
|
@ -2885,6 +2980,7 @@ jobs:
|
|||
category: client
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- install_rust
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
|
|
|
|||
2
Makefile
2
Makefile
|
|
@ -239,7 +239,7 @@ lint-dev: lint-format-changed check-circular-imports check-import-safety
|
|||
# test-linting.yml (Python), test-litellm-ui-build.yml's frontend-lint (dashboard), and
|
||||
# check-ui-api-types.yml (API-type drift), skipping any whose files you didn't stage.
|
||||
# Not auto-installed as a git hook so it never slows an unrelated human commit.
|
||||
pre-commit:
|
||||
pre-commit: bootstrap
|
||||
./scripts/pre_commit_lint.sh
|
||||
|
||||
# Testing targets
|
||||
|
|
|
|||
|
|
@ -0,0 +1,17 @@
|
|||
-- AlterTable
|
||||
ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
|
||||
|
||||
-- AlterTable
|
||||
ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
|
||||
|
||||
-- AlterTable
|
||||
ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
|
||||
|
||||
-- AlterTable
|
||||
ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
|
||||
|
||||
-- AlterTable
|
||||
ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
|
||||
|
||||
-- AlterTable
|
||||
ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
|
||||
|
|
@ -748,6 +748,7 @@ model LiteLLM_DailyUserSpend {
|
|||
compression_saved_tokens BigInt @default(0)
|
||||
compression_savings_spend Float @default(0.0)
|
||||
prompt_caching_savings_spend Float @default(0.0)
|
||||
autorouter_savings_spend Float @default(0.0)
|
||||
spend Float @default(0.0)
|
||||
api_requests BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
|
|
@ -782,6 +783,7 @@ model LiteLLM_DailyOrganizationSpend {
|
|||
compression_saved_tokens BigInt @default(0)
|
||||
compression_savings_spend Float @default(0.0)
|
||||
prompt_caching_savings_spend Float @default(0.0)
|
||||
autorouter_savings_spend Float @default(0.0)
|
||||
spend Float @default(0.0)
|
||||
api_requests BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
|
|
@ -816,6 +818,7 @@ model LiteLLM_DailyEndUserSpend {
|
|||
compression_saved_tokens BigInt @default(0)
|
||||
compression_savings_spend Float @default(0.0)
|
||||
prompt_caching_savings_spend Float @default(0.0)
|
||||
autorouter_savings_spend Float @default(0.0)
|
||||
spend Float @default(0.0)
|
||||
api_requests BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
|
|
@ -849,6 +852,7 @@ model LiteLLM_DailyAgentSpend {
|
|||
compression_saved_tokens BigInt @default(0)
|
||||
compression_savings_spend Float @default(0.0)
|
||||
prompt_caching_savings_spend Float @default(0.0)
|
||||
autorouter_savings_spend Float @default(0.0)
|
||||
spend Float @default(0.0)
|
||||
api_requests BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
|
|
@ -882,6 +886,7 @@ model LiteLLM_DailyTeamSpend {
|
|||
compression_saved_tokens BigInt @default(0)
|
||||
compression_savings_spend Float @default(0.0)
|
||||
prompt_caching_savings_spend Float @default(0.0)
|
||||
autorouter_savings_spend Float @default(0.0)
|
||||
spend Float @default(0.0)
|
||||
api_requests BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
|
|
@ -917,6 +922,7 @@ model LiteLLM_DailyTagSpend {
|
|||
compression_saved_tokens BigInt @default(0)
|
||||
compression_savings_spend Float @default(0.0)
|
||||
prompt_caching_savings_spend Float @default(0.0)
|
||||
autorouter_savings_spend Float @default(0.0)
|
||||
spend Float @default(0.0)
|
||||
api_requests BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
|
|
|
|||
|
|
@ -264,6 +264,7 @@ databricks_key: Optional[str] = None
|
|||
openai_like_key: Optional[str] = None
|
||||
azure_key: Optional[str] = None
|
||||
anthropic_key: Optional[str] = None
|
||||
autorouter_savings_baseline_model: Optional[str] = None
|
||||
replicate_key: Optional[str] = None
|
||||
bytez_key: Optional[str] = None
|
||||
gdc_key: Optional[str] = None
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ from typing import Any, cast
|
|||
# Import all the data structures that define what can be lazy-loaded
|
||||
# These are just lists of names and maps of where to find them
|
||||
from ._lazy_imports_registry import (
|
||||
# Import maps
|
||||
_BEDROCK_TYPES_IMPORT_MAP,
|
||||
_CACHING_IMPORT_MAP,
|
||||
_COST_CALCULATOR_IMPORT_MAP,
|
||||
|
|
@ -33,12 +34,11 @@ from ._lazy_imports_registry import (
|
|||
_TOKEN_COUNTER_IMPORT_MAP,
|
||||
_TYPES_IMPORT_MAP,
|
||||
_TYPES_UTILS_IMPORT_MAP,
|
||||
# Import maps
|
||||
_UTILS_IMPORT_MAP,
|
||||
_UTILS_MODULE_IMPORT_MAP,
|
||||
# Name tuples
|
||||
BEDROCK_TYPES_NAMES,
|
||||
CACHING_NAMES,
|
||||
# Name tuples
|
||||
COST_CALCULATOR_NAMES,
|
||||
DOTPROMPT_NAMES,
|
||||
HTTP_HANDLER_NAMES,
|
||||
|
|
|
|||
|
|
@ -651,7 +651,9 @@ def get_redis_async_client(
|
|||
if arg in args:
|
||||
url_kwargs[arg] = redis_kwargs[arg]
|
||||
else:
|
||||
verbose_logger.debug(f"REDIS: ignoring argument: {arg}. Not an allowed async_redis.Redis.from_url arg.")
|
||||
verbose_logger.debug(
|
||||
"REDIS: ignoring argument: %s. Not an allowed async_redis.Redis.from_url arg.", arg
|
||||
)
|
||||
return async_redis.Redis.from_url(**url_kwargs)
|
||||
|
||||
# Check for Redis Sentinel
|
||||
|
|
@ -805,6 +807,6 @@ def _pretty_print_redis_config(redis_kwargs: dict) -> None:
|
|||
# Fallback to simple logging if rich is not available
|
||||
masker = SensitiveDataMasker()
|
||||
masked_redis_kwargs = masker.mask_dict(redis_kwargs)
|
||||
verbose_logger.info(f"Redis configuration: {masked_redis_kwargs}")
|
||||
verbose_logger.info("Redis configuration: %s", masked_redis_kwargs)
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Error pretty printing Redis configuration: {e}")
|
||||
verbose_logger.error("Error pretty printing Redis configuration: %s", e)
|
||||
|
|
|
|||
|
|
@ -148,13 +148,13 @@ class LiteLLMA2ACardResolver(_A2ACardResolver): # type: ignore[misc]
|
|||
last_error = None
|
||||
for path in paths:
|
||||
try:
|
||||
verbose_logger.debug(f"Attempting to fetch agent card from {self.base_url}{path}")
|
||||
verbose_logger.debug("Attempting to fetch agent card from %s%s", self.base_url, path)
|
||||
return await super().get_agent_card(
|
||||
relative_card_path=path,
|
||||
http_kwargs=http_kwargs,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Failed to fetch agent card from {self.base_url}{path}: {e}")
|
||||
verbose_logger.debug("Failed to fetch agent card from %s%s: %s", self.base_url, path, e)
|
||||
last_error = e
|
||||
continue
|
||||
|
||||
|
|
|
|||
|
|
@ -192,9 +192,11 @@ async def handle_a2a_localhost_retry(
|
|||
|
||||
request_type = "streaming " if is_streaming else ""
|
||||
verbose_logger.warning(
|
||||
f"A2A {request_type}request to '{error.localhost_url}' failed: {error.original_error}. "
|
||||
f"Agent card contains localhost/internal URL. "
|
||||
f"Retrying with base_url '{error.base_url}'."
|
||||
"A2A %srequest to '%s' failed: %s. Agent card contains localhost/internal URL. Retrying with base_url '%s'.",
|
||||
request_type,
|
||||
error.localhost_url,
|
||||
error.original_error,
|
||||
error.base_url,
|
||||
)
|
||||
|
||||
# Fix the agent card URL
|
||||
|
|
|
|||
|
|
@ -76,7 +76,7 @@ class A2ACompletionBridgeHandler:
|
|||
)
|
||||
|
||||
if a2a_provider_config is not None:
|
||||
verbose_logger.info(f"A2A: Using provider config for {custom_llm_provider}")
|
||||
verbose_logger.info("A2A: Using provider config for %s", custom_llm_provider)
|
||||
|
||||
return await a2a_provider_config.handle_non_streaming(
|
||||
request_id=request_id,
|
||||
|
|
@ -103,7 +103,7 @@ class A2ACompletionBridgeHandler:
|
|||
else:
|
||||
full_model = model
|
||||
|
||||
verbose_logger.info(f"A2A completion bridge: model={full_model}, api_base={api_base}")
|
||||
verbose_logger.info("A2A completion bridge: model=%s, api_base=%s", full_model, api_base)
|
||||
|
||||
# Build completion params dict
|
||||
completion_params: dict[str, Any] = {
|
||||
|
|
@ -143,7 +143,7 @@ class A2ACompletionBridgeHandler:
|
|||
request_id=request_id,
|
||||
)
|
||||
|
||||
verbose_logger.info(f"A2A completion bridge completed: request_id={request_id}")
|
||||
verbose_logger.info("A2A completion bridge completed: request_id=%s", request_id)
|
||||
|
||||
return a2a_response
|
||||
|
||||
|
|
@ -185,7 +185,7 @@ class A2ACompletionBridgeHandler:
|
|||
)
|
||||
|
||||
if a2a_provider_config is not None:
|
||||
verbose_logger.info(f"A2A: Using provider config for {custom_llm_provider} (streaming)")
|
||||
verbose_logger.info("A2A: Using provider config for %s (streaming)", custom_llm_provider)
|
||||
|
||||
async for chunk in a2a_provider_config.handle_streaming(
|
||||
request_id=request_id,
|
||||
|
|
@ -221,7 +221,7 @@ class A2ACompletionBridgeHandler:
|
|||
else:
|
||||
full_model = model
|
||||
|
||||
verbose_logger.info(f"A2A completion bridge streaming: model={full_model}, api_base={api_base}")
|
||||
verbose_logger.info("A2A completion bridge streaming: model=%s, api_base=%s", full_model, api_base)
|
||||
|
||||
# Build completion params dict
|
||||
completion_params: dict[str, Any] = {
|
||||
|
|
@ -300,7 +300,9 @@ class A2ACompletionBridgeHandler:
|
|||
)
|
||||
yield completed_event
|
||||
|
||||
verbose_logger.info(f"A2A completion bridge streaming completed: request_id={request_id}, chunks={chunk_count}")
|
||||
verbose_logger.info(
|
||||
"A2A completion bridge streaming completed: request_id=%s, chunks=%s", request_id, chunk_count
|
||||
)
|
||||
|
||||
|
||||
# Convenience functions that delegate to the class methods
|
||||
|
|
|
|||
|
|
@ -109,7 +109,7 @@ class A2ACompletionBridgeTransformation:
|
|||
extra_body = {**extra_body, "metadata": merged_metadata}
|
||||
completion_params["extra_body"] = extra_body
|
||||
|
||||
verbose_logger.debug(f"A2A -> completion forward metadata keys={list(forward_metadata.keys())}")
|
||||
verbose_logger.debug("A2A -> completion forward metadata keys=%s", list(forward_metadata.keys()))
|
||||
|
||||
@staticmethod
|
||||
def a2a_message_to_openai_messages(
|
||||
|
|
@ -145,7 +145,9 @@ class A2ACompletionBridgeTransformation:
|
|||
# once at run level via extra_body.metadata (LangGraph POST /runs/wait shape).
|
||||
openai_message: dict[str, Any] = {"role": openai_role, "content": content}
|
||||
|
||||
verbose_logger.debug(f"A2A -> OpenAI transform: role={role} -> {openai_role}, content_length={len(content)}")
|
||||
verbose_logger.debug(
|
||||
"A2A -> OpenAI transform: role=%s -> %s, content_length=%s", role, openai_role, len(content)
|
||||
)
|
||||
|
||||
return [openai_message]
|
||||
|
||||
|
|
@ -186,7 +188,7 @@ class A2ACompletionBridgeTransformation:
|
|||
"result": a2a_message,
|
||||
}
|
||||
|
||||
verbose_logger.debug(f"OpenAI -> A2A transform: content_length={len(content)}")
|
||||
verbose_logger.debug("OpenAI -> A2A transform: content_length=%s", len(content))
|
||||
|
||||
return a2a_response
|
||||
|
||||
|
|
|
|||
|
|
@ -204,7 +204,7 @@ async def _send_message_via_completion_bridge(
|
|||
|
||||
Requires request; api_base is optional for providers that derive endpoint from model.
|
||||
"""
|
||||
verbose_logger.info(f"A2A using completion bridge: provider={custom_llm_provider}, api_base={api_base}")
|
||||
verbose_logger.info("A2A using completion bridge: provider=%s, api_base=%s", custom_llm_provider, api_base)
|
||||
|
||||
from litellm.a2a_protocol.litellm_completion_bridge.handler import (
|
||||
A2ACompletionBridgeHandler,
|
||||
|
|
@ -463,7 +463,7 @@ async def asend_message(
|
|||
|
||||
agent_name = _get_a2a_model_info(a2a_client, kwargs)
|
||||
|
||||
verbose_logger.info(f"A2A send_message request_id={request.id}, agent={agent_name}")
|
||||
verbose_logger.info("A2A send_message request_id=%s, agent=%s", request.id, agent_name)
|
||||
|
||||
# Get agent card URL for localhost retry logic
|
||||
agent_card = _get_a2a_client_agent_card(a2a_client)
|
||||
|
|
@ -478,7 +478,7 @@ async def asend_message(
|
|||
agent_name=agent_name,
|
||||
)
|
||||
|
||||
verbose_logger.info(f"A2A send_message completed, request_id={request.id}")
|
||||
verbose_logger.info("A2A send_message completed, request_id=%s", request.id)
|
||||
|
||||
# Wrap in LiteLLM response type for _hidden_params support
|
||||
response = LiteLLMSendMessageResponse.from_a2a_response(a2a_response, request_id=str(request.id))
|
||||
|
|
@ -640,7 +640,7 @@ async def asend_message_streaming(
|
|||
raise ValueError("request is required for completion bridge")
|
||||
# api_base is optional for providers that derive endpoint from model (e.g., bedrock/agentcore)
|
||||
|
||||
verbose_logger.info(f"A2A streaming using completion bridge: provider={custom_llm_provider}")
|
||||
verbose_logger.info("A2A streaming using completion bridge: provider=%s", custom_llm_provider)
|
||||
|
||||
from litellm.a2a_protocol.litellm_completion_bridge.handler import (
|
||||
A2ACompletionBridgeHandler,
|
||||
|
|
@ -697,7 +697,7 @@ async def asend_message_streaming(
|
|||
proxy_server_request=proxy_server_request,
|
||||
)
|
||||
|
||||
verbose_logger.info(f"A2A send_message_streaming request_id={request.id}, agent={agent_name}")
|
||||
verbose_logger.info("A2A send_message_streaming request_id=%s, agent=%s", request.id, agent_name)
|
||||
|
||||
agent_card = _get_a2a_client_agent_card(a2a_client)
|
||||
card_url = get_agent_card_url(agent_card) if agent_card else None
|
||||
|
|
@ -759,7 +759,7 @@ async def create_a2a_client(
|
|||
"The 'a2a' package is required for A2A agent invocation. Install it with: pip install a2a-sdk"
|
||||
)
|
||||
|
||||
verbose_logger.info(f"Creating A2A client for {base_url}")
|
||||
verbose_logger.info("Creating A2A client for %s", base_url)
|
||||
|
||||
# Use get_async_httpx_client with per-agent params so that different agents
|
||||
# (with different extra_headers) get separate cached clients. The params
|
||||
|
|
@ -781,7 +781,7 @@ async def create_a2a_client(
|
|||
httpx_client = _async_handler.client
|
||||
if extra_headers:
|
||||
httpx_client.headers.update(extra_headers)
|
||||
verbose_proxy_logger.debug(f"A2A client created with extra_headers={list(extra_headers.keys())}")
|
||||
verbose_proxy_logger.debug("A2A client created with extra_headers=%s", list(extra_headers.keys()))
|
||||
|
||||
a2a_client = await create_client( # pyright: ignore[reportOptionalCall]
|
||||
base_url,
|
||||
|
|
@ -798,7 +798,7 @@ async def create_a2a_client(
|
|||
if agent_card is not None:
|
||||
a2a_client._litellm_agent_card = agent_card # type: ignore[attr-defined]
|
||||
|
||||
verbose_logger.info(f"A2A client created for {base_url}")
|
||||
verbose_logger.info("A2A client created for %s", base_url)
|
||||
|
||||
return a2a_client
|
||||
|
||||
|
|
@ -824,7 +824,7 @@ async def aget_agent_card(
|
|||
"The 'a2a' package is required for A2A agent invocation. Install it with: pip install a2a-sdk"
|
||||
)
|
||||
|
||||
verbose_logger.info(f"Fetching agent card from {base_url}")
|
||||
verbose_logger.info("Fetching agent card from %s", base_url)
|
||||
|
||||
# Use LiteLLM's cached httpx client
|
||||
http_handler = get_async_httpx_client(
|
||||
|
|
@ -839,5 +839,5 @@ async def aget_agent_card(
|
|||
)
|
||||
agent_card = await resolver.get_agent_card()
|
||||
|
||||
verbose_logger.info(f"Fetched agent card: {agent_card.name if hasattr(agent_card, 'name') else 'unknown'}")
|
||||
verbose_logger.info("Fetched agent card: %s", agent_card.name if hasattr(agent_card, "name") else "unknown")
|
||||
return agent_card
|
||||
|
|
|
|||
|
|
@ -53,7 +53,7 @@ class BedrockAgentCoreA2AHandler:
|
|||
agent_extra_headers=agent_extra_headers,
|
||||
)
|
||||
|
||||
verbose_logger.info(f"BedrockAgentCore A2A: Sending non-streaming request to {url}")
|
||||
verbose_logger.info("BedrockAgentCore A2A: Sending non-streaming request to %s", url)
|
||||
|
||||
client = get_async_httpx_client(
|
||||
llm_provider=cast(Any, httpxSpecialProvider.A2AProvider),
|
||||
|
|
@ -67,7 +67,7 @@ class BedrockAgentCoreA2AHandler:
|
|||
response_data = response.json()
|
||||
|
||||
if "error" in response_data:
|
||||
verbose_logger.warning(f"BedrockAgentCore A2A: Agent returned error: {response_data['error']}")
|
||||
verbose_logger.warning("BedrockAgentCore A2A: Agent returned error: %s", response_data["error"])
|
||||
|
||||
return response_data
|
||||
|
||||
|
|
@ -100,7 +100,7 @@ class BedrockAgentCoreA2AHandler:
|
|||
agent_extra_headers=agent_extra_headers,
|
||||
)
|
||||
|
||||
verbose_logger.info(f"BedrockAgentCore A2A: Sending streaming request to {url}")
|
||||
verbose_logger.info("BedrockAgentCore A2A: Sending streaming request to %s", url)
|
||||
|
||||
client = get_async_httpx_client(
|
||||
llm_provider=cast(Any, httpxSpecialProvider.A2AProvider),
|
||||
|
|
|
|||
|
|
@ -195,5 +195,5 @@ class BedrockAgentCoreA2ATransformation:
|
|||
event = json.loads(data_str)
|
||||
yield event
|
||||
except json.JSONDecodeError:
|
||||
verbose_logger.debug(f"BedrockAgentCore A2A: Skipping non-JSON SSE line: {data_str[:100]}")
|
||||
verbose_logger.debug("BedrockAgentCore A2A: Skipping non-JSON SSE line: %s", data_str[:100])
|
||||
continue
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ class PydanticAIHandler:
|
|||
"""
|
||||
if api_base is None:
|
||||
raise ValueError("api_base is required for Pydantic AI agents")
|
||||
verbose_logger.info(f"Pydantic AI: Routing to Pydantic AI agent at {api_base}")
|
||||
verbose_logger.info("Pydantic AI: Routing to Pydantic AI agent at %s", api_base)
|
||||
|
||||
# Send request directly to Pydantic AI agent
|
||||
response_data = await PydanticAITransformation.send_non_streaming_request(
|
||||
|
|
@ -92,7 +92,7 @@ class PydanticAIHandler:
|
|||
"""
|
||||
if api_base is None:
|
||||
raise ValueError("api_base is required for Pydantic AI agents")
|
||||
verbose_logger.info(f"Pydantic AI: Faking streaming for Pydantic AI agent at {api_base}")
|
||||
verbose_logger.info("Pydantic AI: Faking streaming for Pydantic AI agent at %s", api_base)
|
||||
|
||||
# Get raw task response first (not the transformed A2A format)
|
||||
raw_response = await PydanticAITransformation.send_and_get_raw_response(
|
||||
|
|
|
|||
|
|
@ -118,7 +118,7 @@ class PydanticAITransformation:
|
|||
status = result.get("status", {})
|
||||
state = status.get("state", "")
|
||||
|
||||
verbose_logger.debug(f"Pydantic AI: Poll attempt {attempt + 1}/{max_attempts}, state={state}")
|
||||
verbose_logger.debug("Pydantic AI: Poll attempt %s/%s, state=%s", attempt + 1, max_attempts, state)
|
||||
|
||||
if state == "completed":
|
||||
return poll_data
|
||||
|
|
@ -173,7 +173,7 @@ class PydanticAITransformation:
|
|||
# FastA2A uses root endpoint (/) not /messages
|
||||
endpoint = api_base.rstrip("/")
|
||||
|
||||
verbose_logger.info(f"Pydantic AI: Sending non-streaming request to {endpoint}")
|
||||
verbose_logger.info("Pydantic AI: Sending non-streaming request to %s", endpoint)
|
||||
|
||||
# Send request to Pydantic AI agent using shared async HTTP client
|
||||
client = get_async_httpx_client(
|
||||
|
|
@ -200,7 +200,7 @@ class PydanticAITransformation:
|
|||
# Need to poll for completion
|
||||
task_id = result.get("id")
|
||||
if task_id:
|
||||
verbose_logger.info(f"Pydantic AI: Task {task_id} submitted, polling for completion...")
|
||||
verbose_logger.info("Pydantic AI: Task %s submitted, polling for completion...", task_id)
|
||||
response_data = await PydanticAITransformation._poll_for_completion(
|
||||
client=client,
|
||||
endpoint=endpoint,
|
||||
|
|
@ -209,7 +209,7 @@ class PydanticAITransformation:
|
|||
agent_extra_headers=agent_extra_headers,
|
||||
)
|
||||
|
||||
verbose_logger.info(f"Pydantic AI: Received completed response for request_id={request_id}")
|
||||
verbose_logger.info("Pydantic AI: Received completed response for request_id=%s", request_id)
|
||||
|
||||
return response_data
|
||||
|
||||
|
|
@ -518,4 +518,4 @@ class PydanticAITransformation:
|
|||
}
|
||||
yield completed_event
|
||||
|
||||
verbose_logger.info(f"Pydantic AI: Fake streaming completed for request_id={request_id}")
|
||||
verbose_logger.info("Pydantic AI: Fake streaming completed for request_id=%s", request_id)
|
||||
|
|
|
|||
|
|
@ -135,7 +135,7 @@ class WatsonxOrchestrateHandler:
|
|||
response.raise_for_status()
|
||||
result: dict[str, Any] = response.json()
|
||||
status = result.get("status", "")
|
||||
verbose_logger.debug(f"WXO: Poll {attempt + 1}/{max_attempts} run='{run_id}' status='{status}'")
|
||||
verbose_logger.debug("WXO: Poll %s/%s run='%s' status='%s'", attempt + 1, max_attempts, run_id, status)
|
||||
if status in WatsonxOrchestrateTransformation.TERMINAL_STATES:
|
||||
return result
|
||||
|
||||
|
|
@ -297,8 +297,8 @@ class WatsonxOrchestrateHandler:
|
|||
response.raise_for_status()
|
||||
except httpx.TransportError as exc:
|
||||
verbose_logger.warning(
|
||||
f"WXO: Streaming request failed before a run was submitted "
|
||||
f"({exc!r}), falling back to non-streaming + fake streaming",
|
||||
"WXO: Streaming request failed before a run was submitted (%r), falling back to non-streaming + fake streaming",
|
||||
exc,
|
||||
exc_info=True,
|
||||
)
|
||||
result = await WatsonxOrchestrateHandler.handle_non_streaming(
|
||||
|
|
|
|||
|
|
@ -214,4 +214,4 @@ class WatsonxOrchestrateTransformation:
|
|||
},
|
||||
}
|
||||
|
||||
verbose_logger.debug(f"WXO: Fake streaming completed for request_id={request_id}")
|
||||
verbose_logger.debug("WXO: Fake streaming completed for request_id=%s", request_id)
|
||||
|
|
|
|||
|
|
@ -138,13 +138,15 @@ class A2AStreamingIterator:
|
|||
)
|
||||
|
||||
verbose_logger.info(
|
||||
f"A2A streaming completed: prompt_tokens={prompt_tokens}, "
|
||||
f"completion_tokens={completion_tokens}, total_tokens={total_tokens}, "
|
||||
f"response_cost={response_cost}"
|
||||
"A2A streaming completed: prompt_tokens=%s, completion_tokens=%s, total_tokens=%s, response_cost=%s",
|
||||
prompt_tokens,
|
||||
completion_tokens,
|
||||
total_tokens,
|
||||
response_cost,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Error in A2A streaming completion handler: {e}")
|
||||
verbose_logger.debug("Error in A2A streaming completion handler: %s", e)
|
||||
|
||||
def _build_logging_result(self, usage: litellm.Usage) -> dict[str, Any]:
|
||||
"""Build a result dict for logging."""
|
||||
|
|
|
|||
|
|
@ -51,7 +51,7 @@ class GetAnthropicBetaHeadersConfig:
|
|||
)
|
||||
return content
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Failed to load local beta headers config: {e}")
|
||||
verbose_logger.error("Failed to load local beta headers config: %s", e)
|
||||
# Return empty config as fallback
|
||||
return {
|
||||
"anthropic": {},
|
||||
|
|
@ -246,7 +246,9 @@ def filter_and_transform_beta_headers(
|
|||
|
||||
# Check if header is in the mapping
|
||||
if header not in provider_mapping:
|
||||
verbose_logger.debug(f"Dropping unknown beta header '{header}' for provider '{provider}' (not in mapping)")
|
||||
verbose_logger.debug(
|
||||
"Dropping unknown beta header '%s' for provider '%s' (not in mapping)", header, provider
|
||||
)
|
||||
continue
|
||||
|
||||
# Get the mapped header value
|
||||
|
|
@ -254,7 +256,7 @@ def filter_and_transform_beta_headers(
|
|||
|
||||
# Skip if header is unsupported (null value)
|
||||
if mapped_header is None:
|
||||
verbose_logger.debug(f"Dropping unsupported beta header '{header}' for provider '{provider}'")
|
||||
verbose_logger.debug("Dropping unsupported beta header '%s' for provider '%s'", header, provider)
|
||||
continue
|
||||
|
||||
# Add the mapped header
|
||||
|
|
|
|||
|
|
@ -249,7 +249,7 @@ def batch_completion_models_all_responses(*args, **kwargs):
|
|||
if result is not None:
|
||||
responses.append(result)
|
||||
except Exception as e:
|
||||
print_verbose(f"batch_completion_models_all_responses: model request failed: {e!s}")
|
||||
print_verbose(f"batch_completion_models_all_responses: model request failed: {e}")
|
||||
continue
|
||||
|
||||
return responses
|
||||
|
|
|
|||
|
|
@ -258,10 +258,10 @@ async def _fetch_batch_output_file_content(
|
|||
if is_base64_unified_file_id:
|
||||
try:
|
||||
file_id = is_base64_unified_file_id.split("llm_output_file_id,")[1].split(";")[0]
|
||||
verbose_logger.debug(f"Extracted LLM output file ID from unified file ID: {file_id}")
|
||||
verbose_logger.debug("Extracted LLM output file ID from unified file ID: %s", file_id)
|
||||
except (IndexError, AttributeError) as e:
|
||||
verbose_logger.error(
|
||||
f"Failed to extract LLM output file ID from unified file ID: {batch.output_file_id}, error: {e}"
|
||||
"Failed to extract LLM output file ID from unified file ID: %s, error: %s", batch.output_file_id, e
|
||||
)
|
||||
|
||||
# Build kwargs for afile_content with credentials from litellm_params
|
||||
|
|
|
|||
|
|
@ -182,7 +182,7 @@ def create_batch(
|
|||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(
|
||||
f"litellm.batches.main.py::create_batch() - Error inferring custom_llm_provider - {e!s}"
|
||||
"litellm.batches.main.py::create_batch() - Error inferring custom_llm_provider - %s", e
|
||||
)
|
||||
|
||||
_is_async = kwargs.pop("acreate_batch", False) is True
|
||||
|
|
@ -890,7 +890,7 @@ def cancel_batch(
|
|||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(
|
||||
f"litellm.batches.main.py::cancel_batch() - Error inferring custom_llm_provider - {e!s}"
|
||||
"litellm.batches.main.py::cancel_batch() - Error inferring custom_llm_provider - %s", e
|
||||
)
|
||||
optional_params = GenericLiteLLMParams(**kwargs)
|
||||
litellm_params = get_litellm_params(
|
||||
|
|
|
|||
|
|
@ -67,7 +67,10 @@ class AzureBlobCache(BaseCache):
|
|||
cached_response = json.loads(as_str)
|
||||
|
||||
verbose_logger.debug(
|
||||
f"Got Azure Blob Cache: key: {key}, cached_response {cached_response}. Type Response {type(cached_response)}"
|
||||
"Got Azure Blob Cache: key: %s, cached_response %s. Type Response %s",
|
||||
key,
|
||||
cached_response,
|
||||
type(cached_response),
|
||||
)
|
||||
|
||||
return cached_response
|
||||
|
|
@ -84,7 +87,10 @@ class AzureBlobCache(BaseCache):
|
|||
as_str = as_bytes.decode("utf-8")
|
||||
cached_response = json.loads(as_str)
|
||||
verbose_logger.debug(
|
||||
f"Got Azure Blob Cache: key: {key}, cached_response {cached_response}. Type Response {type(cached_response)}"
|
||||
"Got Azure Blob Cache: key: %s, cached_response %s. Type Response %s",
|
||||
key,
|
||||
cached_response,
|
||||
type(cached_response),
|
||||
)
|
||||
return cached_response
|
||||
except ResourceNotFoundError:
|
||||
|
|
|
|||
|
|
@ -353,13 +353,13 @@ class Cache:
|
|||
if param in combined_kwargs:
|
||||
param_value: str | None = self._get_param_value(param, kwargs)
|
||||
if param_value is not None:
|
||||
cache_key += f"{param!s}: {param_value!s}"
|
||||
cache_key += f"{param}: {param_value}"
|
||||
elif param not in litellm_param_kwargs: # check if user passed in optional param - e.g. top_k
|
||||
if litellm.enable_caching_on_provider_specific_optional_params is True: # feature flagged for now
|
||||
if kwargs[param] is None:
|
||||
continue # ignore None params
|
||||
param_value = kwargs[param]
|
||||
cache_key += f"{param!s}: {param_value!s}"
|
||||
cache_key += f"{param}: {param_value}"
|
||||
|
||||
if is_semantic_cache:
|
||||
cache_key += self._get_semantic_cache_tenant_scope(kwargs)
|
||||
|
|
@ -676,7 +676,7 @@ class Cache:
|
|||
cache_key, cached_data, kwargs = self._add_cache_logic(result=result, **kwargs)
|
||||
self.cache.set_cache(cache_key, cached_data, **kwargs)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"LiteLLM Cache: Excepton add_cache: {e!s}")
|
||||
verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e)
|
||||
|
||||
async def async_add_cache(self, result, dynamic_cache_object: BaseCache | None = None, **kwargs):
|
||||
"""
|
||||
|
|
@ -695,7 +695,7 @@ class Cache:
|
|||
else:
|
||||
await self.cache.async_set_cache(cache_key, cached_data, **kwargs)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"LiteLLM Cache: Excepton add_cache: {e!s}")
|
||||
verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e)
|
||||
|
||||
def _convert_to_cached_embedding(
|
||||
self,
|
||||
|
|
@ -874,7 +874,7 @@ class Cache:
|
|||
else:
|
||||
await self.cache.async_set_cache_pipeline(cache_list=cache_list, **kwargs)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"LiteLLM Cache: Excepton add_cache: {e!s}")
|
||||
verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e)
|
||||
|
||||
def should_use_cache(self, **kwargs):
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -271,7 +271,7 @@ class LLMCachingHandler:
|
|||
embedding_all_elements_cache_hit=embedding_all_elements_cache_hit,
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"CACHE RESULT: {cached_result}")
|
||||
verbose_logger.debug("CACHE RESULT: %s", cached_result)
|
||||
return CachingHandlerResponse(
|
||||
cached_result=cached_result,
|
||||
final_embedding_cached_response=final_embedding_cached_response,
|
||||
|
|
|
|||
|
|
@ -147,7 +147,7 @@ class DualCache(BaseCache):
|
|||
|
||||
return result
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"LiteLLM Cache: Excepton async add_cache: {e!s}")
|
||||
verbose_logger.error("LiteLLM Cache: Excepton async add_cache: %s", e)
|
||||
raise e
|
||||
|
||||
def get_cache(
|
||||
|
|
@ -347,7 +347,7 @@ class DualCache(BaseCache):
|
|||
if self.redis_cache is not None and local_only is False:
|
||||
await self.redis_cache.async_set_cache(key, value, **kwargs)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"LiteLLM Cache: Excepton async add_cache: {e!s}")
|
||||
verbose_logger.exception("LiteLLM Cache: Excepton async add_cache: %s", e)
|
||||
|
||||
# async_batch_set_cache
|
||||
async def async_set_cache_pipeline(self, cache_list: list, local_only: bool = False, **kwargs):
|
||||
|
|
@ -366,7 +366,7 @@ class DualCache(BaseCache):
|
|||
cache_list=cache_list, ttl=kwargs.pop("ttl", None), **kwargs
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"LiteLLM Cache: Excepton async add_cache: {e!s}")
|
||||
verbose_logger.exception("LiteLLM Cache: Excepton async add_cache: %s", e)
|
||||
|
||||
async def async_increment_cache(
|
||||
self,
|
||||
|
|
|
|||
276
litellm/caching/evicted_client_closer.py
Normal file
276
litellm/caching/evicted_client_closer.py
Normal file
|
|
@ -0,0 +1,276 @@
|
|||
"""
|
||||
Deferred close of HTTP/SDK clients that the LLM client cache has evicted.
|
||||
|
||||
Eviction only drops the cache's reference to a client. Every OpenAI/Azure SDK
|
||||
client is a reference cycle (each resource namespace holds the client back), so
|
||||
an evicted client and its pooled TCP connections survive until a generational
|
||||
collection runs, which under load is thousands of requests later.
|
||||
|
||||
Closing at eviction time is not an option: a request that was handed the client
|
||||
just before it was evicted is still using it, and closing it underneath that
|
||||
request raises ``RuntimeError: Cannot send a request, as the client has been
|
||||
closed.``
|
||||
|
||||
So an evicted client is closed once two conditions hold. A grace window must
|
||||
have passed since its eviction, which covers a request that holds the client
|
||||
but is momentarily not on the wire, and the client must report no connection in
|
||||
flight. The second condition is what keeps the first honest: a request may run
|
||||
for ``litellm.request_timeout`` seconds, 6000 by default, and a streaming
|
||||
response is bounded only by how long the upstream keeps sending, so no deadline
|
||||
on its own can promise that a request has finished.
|
||||
|
||||
Only clients litellm itself created are closed; a client the caller supplied is
|
||||
left alone because litellm does not own its lifecycle.
|
||||
|
||||
A client that closes synchronously is closed from wherever the cache is next
|
||||
used. One whose close is a coroutine needs the event loop it was evicted on, so
|
||||
it waits for a call from that loop rather than having work scheduled onto a loop
|
||||
it does not belong to. Queued clients are therefore bucketed by what it takes to
|
||||
close them, and each bucket is ordered by deadline, so a reap walks the entries
|
||||
that are due rather than the whole queue.
|
||||
|
||||
The queue holds its clients weakly, so waiting out a grace window never keeps
|
||||
alive anything the collector would have reclaimed first.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import inspect
|
||||
import threading
|
||||
import time
|
||||
import weakref
|
||||
from collections import deque
|
||||
from collections.abc import Awaitable, Callable, Iterator
|
||||
from dataclasses import dataclass, replace
|
||||
|
||||
from litellm.constants import (
|
||||
EVICTED_LLM_CLIENT_CLOSE_GRACE_SECONDS,
|
||||
EVICTED_LLM_CLIENT_CLOSE_MAX_PENDING,
|
||||
)
|
||||
|
||||
_CLOSABLE_ANYWHERE = "closable-anywhere"
|
||||
_CLOSABLE_ON_ANY_LOOP = "closable-on-any-loop"
|
||||
|
||||
_BucketKey = str | int
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _PendingClose:
|
||||
"""A queued close.
|
||||
|
||||
The client is held weakly, so queueing one never keeps alive anything the
|
||||
collector would otherwise have reclaimed first.
|
||||
|
||||
``needs_loop`` is set for a client whose close is a coroutine; those can only
|
||||
be closed from the event loop they were evicted on, recorded in ``loop_id``.
|
||||
A client that closes synchronously carries neither constraint.
|
||||
"""
|
||||
|
||||
client_ref: "weakref.ref[object]"
|
||||
loop_id: int | None
|
||||
needs_loop: bool
|
||||
close_after: float
|
||||
|
||||
|
||||
def _bucket_key(pending: _PendingClose) -> _BucketKey:
|
||||
"""Which reaps can close this entry: any at all, any running a loop, or one loop's."""
|
||||
if not pending.needs_loop:
|
||||
return _CLOSABLE_ANYWHERE
|
||||
if pending.loop_id is None:
|
||||
return _CLOSABLE_ON_ANY_LOOP
|
||||
return pending.loop_id
|
||||
|
||||
|
||||
def _running_loop_id() -> int | None:
|
||||
try:
|
||||
return id(asyncio.get_running_loop())
|
||||
except RuntimeError:
|
||||
return None
|
||||
|
||||
|
||||
def _close_function(client: object) -> Callable[[], object] | None:
|
||||
close_fn: Callable[[], object] | None = getattr(client, "aclose", None) or getattr(client, "close", None)
|
||||
return close_fn
|
||||
|
||||
|
||||
def _transport_of(client: object) -> object:
|
||||
"""The httpx transport behind an SDK wrapper, a litellm handler, or a bare client."""
|
||||
for holder in (getattr(client, "_client", None), getattr(client, "client", None), client):
|
||||
transport: object = getattr(holder, "_transport", None)
|
||||
if transport is not None:
|
||||
return transport
|
||||
return None
|
||||
|
||||
|
||||
def _connection_is_idle(connection: object) -> bool:
|
||||
"""A pooled connection is idle unless it is servicing a request."""
|
||||
is_idle: object = getattr(connection, "is_idle", None)
|
||||
return bool(is_idle()) if callable(is_idle) else True
|
||||
|
||||
|
||||
def _pool_has_busy_connection(transport: object) -> bool | None:
|
||||
"""Whether the httpcore pool behind the transport is servicing a request.
|
||||
|
||||
``None`` when there is no such pool, so the caller can ask the other backend.
|
||||
"""
|
||||
pooled: object = getattr(getattr(transport, "_pool", None), "connections", None)
|
||||
if not isinstance(pooled, (list, tuple)):
|
||||
return None
|
||||
return any(
|
||||
not _connection_is_idle(connection) # pyright: ignore[reportUnknownArgumentType] # untyped pool list
|
||||
for connection in pooled # pyright: ignore[reportUnknownVariableType] # untyped pool list
|
||||
)
|
||||
|
||||
|
||||
def _has_connection_in_flight(client: object) -> bool:
|
||||
"""Whether the client is servicing a request right now.
|
||||
|
||||
Both connection backends litellm uses already account for the connections
|
||||
they have handed out, so this reads the client's own lease accounting rather
|
||||
than inferring it from elapsed time: httpcore reports a non-idle connection
|
||||
for the whole of a response including a stream, and aiohttp holds the
|
||||
connection in ``_acquired`` over the same span.
|
||||
|
||||
A client that cannot answer is reported as idle, which leaves the grace
|
||||
window as the only guard, exactly as it was before this check existed.
|
||||
"""
|
||||
try:
|
||||
transport = _transport_of(client)
|
||||
pooled_busy = _pool_has_busy_connection(transport)
|
||||
if pooled_busy is not None:
|
||||
return pooled_busy
|
||||
session: object = getattr(transport, "client", None)
|
||||
return bool(getattr(getattr(session, "connector", None), "_acquired", None))
|
||||
except Exception: # noqa: BLE001 - a client that cannot report its state is treated as idle
|
||||
return False
|
||||
|
||||
|
||||
async def _close_quietly(closing: Awaitable[object]) -> None:
|
||||
try:
|
||||
await closing
|
||||
except Exception: # noqa: BLE001 - a discarded client's close must never surface to callers
|
||||
pass
|
||||
|
||||
|
||||
class EvictedClientCloser:
|
||||
"""Closes evicted, litellm-owned clients once they are idle and out of grace."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
grace_seconds: float = EVICTED_LLM_CLIENT_CLOSE_GRACE_SECONDS,
|
||||
max_pending: int = EVICTED_LLM_CLIENT_CLOSE_MAX_PENDING,
|
||||
clock: Callable[[], float] = time.monotonic,
|
||||
) -> None:
|
||||
self._grace_seconds = grace_seconds
|
||||
self._max_pending = max_pending
|
||||
self._clock = clock
|
||||
self._owned: weakref.WeakSet[object] = weakref.WeakSet()
|
||||
self._buckets: dict[_BucketKey, deque[_PendingClose]] = {} # mutable-ok: deadline-ordered queues
|
||||
self._pending_count = 0
|
||||
self._queue_lock = threading.Lock() # the cache is reachable from every worker thread's loop
|
||||
self._close_tasks: set[asyncio.Task[None]] = set() # mutable-ok: strong refs to running closes
|
||||
|
||||
def mark_owned(self, client: object) -> None:
|
||||
"""Record that litellm created this client, so it may be closed on eviction."""
|
||||
try:
|
||||
self._owned.add(client)
|
||||
except TypeError:
|
||||
pass # values that cannot be weak-referenced are never litellm clients
|
||||
|
||||
def _is_owned(self, client: object) -> bool:
|
||||
try:
|
||||
return client in self._owned
|
||||
except TypeError:
|
||||
return False # unhashable values are never litellm clients
|
||||
|
||||
def schedule(self, client: object) -> None:
|
||||
"""Queue an evicted client for closing once it is idle and out of grace.
|
||||
|
||||
Past ``max_pending`` the client is left to the collector instead, so a
|
||||
workload that churns the cache cannot grow this queue without bound.
|
||||
Every queued entry comes due within one grace window, so the capacity it
|
||||
occupies is returned within that window rather than held.
|
||||
"""
|
||||
if client is None or not self._is_owned(client):
|
||||
return
|
||||
close_fn = _close_function(client)
|
||||
if close_fn is None:
|
||||
return
|
||||
if self._pending_count >= self._max_pending:
|
||||
return
|
||||
self._enqueue(
|
||||
_PendingClose(
|
||||
client_ref=weakref.ref(client),
|
||||
loop_id=_running_loop_id(),
|
||||
needs_loop=inspect.iscoroutinefunction(close_fn),
|
||||
close_after=self._clock() + self._grace_seconds,
|
||||
)
|
||||
)
|
||||
|
||||
def reap(self) -> None:
|
||||
"""Close every queued client that is due, idle, and closable from here.
|
||||
|
||||
Called from the cache's read path, so the empty-queue exit comes first and
|
||||
the work done past it is proportional to what is due, not to the queue.
|
||||
"""
|
||||
if not self._pending_count:
|
||||
return
|
||||
now = self._clock()
|
||||
for pending in self._take_due(_running_loop_id(), now):
|
||||
client = pending.client_ref()
|
||||
if client is None:
|
||||
continue
|
||||
if _has_connection_in_flight(client):
|
||||
self._enqueue(replace(pending, close_after=now + self._grace_seconds))
|
||||
continue
|
||||
self._close(client)
|
||||
|
||||
@property
|
||||
def pending_count(self) -> int:
|
||||
return self._pending_count
|
||||
|
||||
def _enqueue(self, pending: _PendingClose) -> None:
|
||||
"""Append to the entry's bucket, dropping any dead entries it queues behind.
|
||||
|
||||
Deadlines only ever move forward, so appending keeps each bucket ordered
|
||||
by deadline, and entries whose client the collector already took sit at
|
||||
the front rather than having to be searched for.
|
||||
"""
|
||||
with self._queue_lock:
|
||||
bucket = self._buckets.setdefault(_bucket_key(pending), deque()) # mutable-ok: FIFO by design
|
||||
while bucket and bucket[0].client_ref() is None:
|
||||
bucket.popleft()
|
||||
self._pending_count -= 1
|
||||
bucket.append(pending)
|
||||
self._pending_count += 1
|
||||
|
||||
def _take_due(self, loop_id: int | None, now: float) -> tuple[_PendingClose, ...]:
|
||||
buckets = (_CLOSABLE_ANYWHERE,) if loop_id is None else (_CLOSABLE_ANYWHERE, _CLOSABLE_ON_ANY_LOOP, loop_id)
|
||||
with self._queue_lock:
|
||||
return tuple(pending for key in buckets for pending in self._drain_locked(key, now))
|
||||
|
||||
def _drain_locked(self, key: _BucketKey, now: float) -> Iterator[_PendingClose]:
|
||||
bucket = self._buckets.get(key)
|
||||
if bucket is None:
|
||||
return
|
||||
while bucket and bucket[0].close_after <= now:
|
||||
self._pending_count -= 1
|
||||
yield bucket.popleft()
|
||||
if not bucket:
|
||||
del self._buckets[key]
|
||||
|
||||
def _close(self, client: object) -> None:
|
||||
close_fn = _close_function(client)
|
||||
if close_fn is None:
|
||||
return
|
||||
try:
|
||||
closing = close_fn()
|
||||
except Exception: # noqa: BLE001 - a discarded client's close must never surface to callers
|
||||
return
|
||||
if not inspect.isawaitable(closing):
|
||||
return
|
||||
task = asyncio.get_running_loop().create_task(_close_quietly(closing))
|
||||
self._close_tasks.add(task)
|
||||
task.add_done_callback(self._close_tasks.discard)
|
||||
|
||||
|
||||
default_evicted_client_closer = EvictedClientCloser()
|
||||
|
|
@ -71,12 +71,15 @@ class GCSCache(BaseCache):
|
|||
if response.status_code == 200:
|
||||
cached_response = json.loads(response.text)
|
||||
verbose_logger.debug(
|
||||
f"Got GCS Cache: key: {key}, cached_response {cached_response}. Type Response {type(cached_response)}"
|
||||
"Got GCS Cache: key: %s, cached_response %s. Type Response %s",
|
||||
key,
|
||||
cached_response,
|
||||
type(cached_response),
|
||||
)
|
||||
return cached_response
|
||||
return None
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"GCS Caching: get_cache() - Got exception from GCS: {e}")
|
||||
verbose_logger.error("GCS Caching: get_cache() - Got exception from GCS: %s", e)
|
||||
|
||||
async def async_get_cache(self, key, **kwargs):
|
||||
try:
|
||||
|
|
@ -89,7 +92,7 @@ class GCSCache(BaseCache):
|
|||
return json.loads(response.text)
|
||||
return None
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"GCS Caching: async_get_cache() - Got exception from GCS: {e}")
|
||||
verbose_logger.error("GCS Caching: async_get_cache() - Got exception from GCS: %s", e)
|
||||
|
||||
def flush_cache(self):
|
||||
pass
|
||||
|
|
|
|||
|
|
@ -4,21 +4,44 @@ Add the event loop to the cache key, to prevent event loop closed errors.
|
|||
|
||||
import asyncio
|
||||
|
||||
from .evicted_client_closer import EvictedClientCloser, default_evicted_client_closer
|
||||
from .in_memory_cache import InMemoryCache
|
||||
|
||||
|
||||
class LLMClientCache(InMemoryCache):
|
||||
"""Cache for LLM HTTP clients (OpenAI, Azure, httpx, etc.).
|
||||
|
||||
IMPORTANT: This cache intentionally does NOT close clients on eviction.
|
||||
Evicted clients may still be in use by in-flight requests. Closing them
|
||||
eagerly causes ``RuntimeError: Cannot send a request, as the client has
|
||||
been closed.`` errors in production after the TTL (1 hour) expires.
|
||||
An evicted client is never closed on the spot: a request handed the client
|
||||
just before eviction is still using it, and closing it there raises
|
||||
``RuntimeError: Cannot send a request, as the client has been closed.``
|
||||
|
||||
Clients that are no longer referenced will be garbage-collected normally.
|
||||
For explicit shutdown cleanup, use ``close_litellm_async_clients()``.
|
||||
Nor can eviction be left to rely on garbage collection. The SDK clients are
|
||||
reference cycles, so an evicted client and its open TCP connections survive
|
||||
until a generational collection runs. Instead a client litellm created is
|
||||
handed to ``EvictedClientCloser``, which closes it once a grace window has
|
||||
passed. Clients the caller supplied are left untouched.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
max_size_in_memory: int | None = 200,
|
||||
default_ttl: int | None = 600,
|
||||
max_size_per_item: int | None = 1024,
|
||||
evicted_client_closer: EvictedClientCloser | None = None,
|
||||
):
|
||||
super().__init__(
|
||||
max_size_in_memory=max_size_in_memory,
|
||||
default_ttl=default_ttl,
|
||||
max_size_per_item=max_size_per_item,
|
||||
)
|
||||
self.evicted_client_closer = evicted_client_closer or default_evicted_client_closer
|
||||
|
||||
def _remove_key(self, key: str) -> None:
|
||||
evicted: object = self.cache_dict.get(key)
|
||||
super()._remove_key(key)
|
||||
self.evicted_client_closer.schedule(evicted)
|
||||
self.evicted_client_closer.reap()
|
||||
|
||||
def update_cache_key_with_event_loop(self, key):
|
||||
"""
|
||||
Add the event loop to the cache key, to prevent event loop closed errors.
|
||||
|
|
@ -31,16 +54,22 @@ class LLMClientCache(InMemoryCache):
|
|||
except RuntimeError: # handle no current running event loop
|
||||
return key
|
||||
|
||||
def set_cache(self, key, value, **kwargs):
|
||||
def set_cache(self, key: str, value: object, litellm_owned_client: bool = False, **kwargs):
|
||||
"""``litellm_owned_client`` marks a client litellm built, so it may be closed once evicted."""
|
||||
if litellm_owned_client:
|
||||
self.evicted_client_closer.mark_owned(value)
|
||||
key = self.update_cache_key_with_event_loop(key)
|
||||
return super().set_cache(key, value, **kwargs)
|
||||
|
||||
async def async_set_cache(self, key, value, **kwargs):
|
||||
async def async_set_cache(self, key: str, value: object, litellm_owned_client: bool = False, **kwargs):
|
||||
if litellm_owned_client:
|
||||
self.evicted_client_closer.mark_owned(value)
|
||||
key = self.update_cache_key_with_event_loop(key)
|
||||
return await super().async_set_cache(key, value, **kwargs)
|
||||
|
||||
def get_cache(self, key, **kwargs):
|
||||
key = self.update_cache_key_with_event_loop(key)
|
||||
self.evicted_client_closer.reap()
|
||||
|
||||
return super().get_cache(key, **kwargs)
|
||||
|
||||
|
|
|
|||
|
|
@ -178,7 +178,7 @@ class QdrantSemanticCache(BaseCache):
|
|||
if response.status_code not in (200, 201):
|
||||
print_verbose(f"Qdrant semantic-cache could not create cache-key payload index: {response.text}")
|
||||
except Exception as exc:
|
||||
print_verbose(f"Qdrant semantic-cache could not create cache-key payload index: {exc!s}")
|
||||
print_verbose(f"Qdrant semantic-cache could not create cache-key payload index: {exc}")
|
||||
|
||||
def _payload_matches_cache_key(self, payload: dict, key: str) -> bool:
|
||||
# Pre-isolation points stored only prompt + response with no cache-key
|
||||
|
|
|
|||
|
|
@ -346,7 +346,8 @@ class RedisCache(BaseCache):
|
|||
verbose_logger.debug("Ignoring async redis ping. No running event loop.")
|
||||
else:
|
||||
verbose_logger.error(
|
||||
f"Error connecting to Async Redis client - {e!s}",
|
||||
"Error connecting to Async Redis client - %s",
|
||||
e,
|
||||
extra={"error": str(e)},
|
||||
)
|
||||
self._handle_async_ping_error(e)
|
||||
|
|
@ -483,7 +484,7 @@ class RedisCache(BaseCache):
|
|||
)
|
||||
except Exception as e:
|
||||
# NON blocking - notify users Redis is throwing an exception
|
||||
print_verbose(f"litellm.caching.caching: set() - Got exception from REDIS : {e!s}")
|
||||
print_verbose(f"litellm.caching.caching: set() - Got exception from REDIS : {e}")
|
||||
|
||||
def increment_cache(self, key, value: int, ttl: float | None = None, **kwargs) -> int:
|
||||
_redis_client = self.redis_client
|
||||
|
|
@ -1139,7 +1140,7 @@ class RedisCache(BaseCache):
|
|||
|
||||
return decoded_results
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Error occurred in batch get cache - {e!s}")
|
||||
verbose_logger.error("Error occurred in batch get cache - %s", e)
|
||||
return key_value_dict
|
||||
|
||||
@_redis_circuit_breaker_guard
|
||||
|
|
@ -1185,7 +1186,7 @@ class RedisCache(BaseCache):
|
|||
event_metadata={"key": key},
|
||||
)
|
||||
)
|
||||
print_verbose(f"litellm.caching.caching: async get() - Got exception from REDIS: {e!s}")
|
||||
print_verbose(f"litellm.caching.caching: async get() - Got exception from REDIS: {e}")
|
||||
_record_swallowed_redis_failure(self._circuit_breaker, e)
|
||||
|
||||
@_redis_circuit_breaker_guard
|
||||
|
|
@ -1257,7 +1258,7 @@ class RedisCache(BaseCache):
|
|||
parent_otel_span=parent_otel_span,
|
||||
)
|
||||
)
|
||||
verbose_logger.error(f"Error occurred in async batch get cache - {e!s}")
|
||||
verbose_logger.error("Error occurred in async batch get cache - %s", e)
|
||||
_record_swallowed_redis_failure(self._circuit_breaker, e)
|
||||
return key_value_dict
|
||||
|
||||
|
|
@ -1292,7 +1293,7 @@ class RedisCache(BaseCache):
|
|||
error=e,
|
||||
call_type=f"sync_ping <- {_get_call_stack_info()}",
|
||||
)
|
||||
verbose_logger.error(f"LiteLLM Redis Cache PING: - Got exception from REDIS : {e!s}")
|
||||
verbose_logger.error("LiteLLM Redis Cache PING: - Got exception from REDIS : %s", e)
|
||||
raise e
|
||||
|
||||
async def ping(self) -> bool:
|
||||
|
|
@ -1326,7 +1327,7 @@ class RedisCache(BaseCache):
|
|||
call_type=f"async_ping <- {_get_call_stack_info()}",
|
||||
)
|
||||
)
|
||||
verbose_logger.error(f"LiteLLM Redis Cache PING: - Got exception from REDIS : {e!s}")
|
||||
verbose_logger.error("LiteLLM Redis Cache PING: - Got exception from REDIS : %s", e)
|
||||
raise e
|
||||
|
||||
@_redis_circuit_breaker_guard
|
||||
|
|
@ -1388,10 +1389,10 @@ class RedisCache(BaseCache):
|
|||
else:
|
||||
return {"status": "failed", "message": "Redis ping returned False"}
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Redis connection test failed: {e!s}")
|
||||
verbose_logger.error("Redis connection test failed: %s", e)
|
||||
return {
|
||||
"status": "failed",
|
||||
"message": f"Redis connection failed: {e!s}",
|
||||
"message": f"Redis connection failed: {e}",
|
||||
"error": str(e),
|
||||
}
|
||||
|
||||
|
|
@ -1426,7 +1427,7 @@ class RedisCache(BaseCache):
|
|||
# Execute the pipeline and return results
|
||||
results = await pipe.execute()
|
||||
# only return float values
|
||||
verbose_logger.debug(f"Increment ASYNC Redis Cache PIPELINE: results: {results}")
|
||||
verbose_logger.debug("Increment ASYNC Redis Cache PIPELINE: results: %s", results)
|
||||
return [r for r in results if isinstance(r, float)]
|
||||
|
||||
@_redis_circuit_breaker_guard
|
||||
|
|
@ -1513,7 +1514,7 @@ class RedisCache(BaseCache):
|
|||
return None
|
||||
return ttl
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Redis TTL Error: {e}")
|
||||
verbose_logger.debug("Redis TTL Error: %s", e)
|
||||
_record_swallowed_redis_failure(self._circuit_breaker, e)
|
||||
return None
|
||||
|
||||
|
|
@ -1565,7 +1566,7 @@ class RedisCache(BaseCache):
|
|||
call_type=f"async_rpush <- {_get_call_stack_info()}",
|
||||
)
|
||||
)
|
||||
verbose_logger.error(f"LiteLLM Redis Cache RPUSH: - Got exception from REDIS : {e!s}")
|
||||
verbose_logger.error("LiteLLM Redis Cache RPUSH: - Got exception from REDIS : %s", e)
|
||||
raise e
|
||||
|
||||
async def _pipeline_rpush_helper(
|
||||
|
|
@ -1711,7 +1712,7 @@ class RedisCache(BaseCache):
|
|||
call_type=f"async_lpop <- {_get_call_stack_info()}",
|
||||
)
|
||||
)
|
||||
verbose_logger.error(f"LiteLLM Redis Cache LPOP: - Got exception from REDIS : {e!s}")
|
||||
verbose_logger.error("LiteLLM Redis Cache LPOP: - Got exception from REDIS : %s", e)
|
||||
raise e
|
||||
|
||||
async def _pipeline_lpop_helper(
|
||||
|
|
|
|||
|
|
@ -100,9 +100,9 @@ class RedisClusterCache(RedisCache):
|
|||
except Exception as e:
|
||||
from litellm._logging import verbose_logger
|
||||
|
||||
verbose_logger.error(f"Redis Cluster connection test failed: {e!s}")
|
||||
verbose_logger.error("Redis Cluster connection test failed: %s", e)
|
||||
return {
|
||||
"status": "failed",
|
||||
"message": f"Redis Cluster connection failed: {e!s}",
|
||||
"message": f"Redis Cluster connection failed: {e}",
|
||||
"error": str(e),
|
||||
}
|
||||
|
|
|
|||
|
|
@ -138,7 +138,7 @@ class RedisSemanticCache(BaseCache):
|
|||
cache_vectorizer=cache_vectorizer,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Redis semantic-cache index build failed: {e}")
|
||||
verbose_logger.error("Redis semantic-cache index build failed: %s", e)
|
||||
raise
|
||||
|
||||
@classmethod
|
||||
|
|
@ -364,7 +364,7 @@ class RedisSemanticCache(BaseCache):
|
|||
try:
|
||||
cached_response = ast.literal_eval(cached_response)
|
||||
except (ValueError, SyntaxError) as e:
|
||||
print_verbose(f"Error parsing cached response: {e!s}")
|
||||
print_verbose(f"Error parsing cached response: {e}")
|
||||
return None
|
||||
|
||||
return cached_response
|
||||
|
|
@ -403,7 +403,7 @@ class RedisSemanticCache(BaseCache):
|
|||
store_kwargs["ttl"] = int(ttl)
|
||||
self.llmcache.store(prompt, value_str, **store_kwargs)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error setting {value_str or value} in the Redis semantic cache: {e!s}")
|
||||
print_verbose(f"Error setting {value_str or value} in the Redis semantic cache: {e}")
|
||||
|
||||
def get_cache(self, key: str, **kwargs) -> Any:
|
||||
"""
|
||||
|
|
@ -468,7 +468,7 @@ class RedisSemanticCache(BaseCache):
|
|||
|
||||
return self._get_cache_logic(cached_response=cached_response)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error retrieving from Redis semantic cache: {e!s}")
|
||||
print_verbose(f"Error retrieving from Redis semantic cache: {e}")
|
||||
kwargs.setdefault("metadata", {})["semantic-similarity"] = 0.0
|
||||
|
||||
async def _get_async_embedding(self, prompt: str, metadata: dict[str, Any] | None = None) -> list[float]:
|
||||
|
|
@ -505,8 +505,8 @@ class RedisSemanticCache(BaseCache):
|
|||
)
|
||||
return embedding_response["data"][0]["embedding"]
|
||||
except Exception as e:
|
||||
print_verbose(f"Error generating async embedding: {e!s}")
|
||||
raise ValueError(f"Failed to generate embedding: {e!s}") from e
|
||||
print_verbose(f"Error generating async embedding: {e}")
|
||||
raise ValueError(f"Failed to generate embedding: {e}") from e
|
||||
|
||||
async def async_set_cache(self, key: str, value: Any, **kwargs) -> None:
|
||||
"""
|
||||
|
|
@ -546,7 +546,7 @@ class RedisSemanticCache(BaseCache):
|
|||
**store_kwargs,
|
||||
)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error in async_set_cache: {e!s}")
|
||||
print_verbose(f"Error in async_set_cache: {e}")
|
||||
|
||||
async def async_get_cache(self, key: str, **kwargs) -> Any:
|
||||
"""
|
||||
|
|
@ -612,7 +612,7 @@ class RedisSemanticCache(BaseCache):
|
|||
|
||||
return self._get_cache_logic(cached_response=cached_response)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error in async_get_cache: {e!s}")
|
||||
print_verbose(f"Error in async_get_cache: {e}")
|
||||
kwargs.setdefault("metadata", {})["semantic-similarity"] = 0.0
|
||||
|
||||
async def _index_info(self) -> dict[str, Any]:
|
||||
|
|
@ -639,4 +639,4 @@ class RedisSemanticCache(BaseCache):
|
|||
tasks.append(self.async_set_cache(val[0], val[1], **kwargs))
|
||||
await asyncio.gather(*tasks)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error in async_set_cache_pipeline: {e!s}")
|
||||
print_verbose(f"Error in async_set_cache_pipeline: {e}")
|
||||
|
|
|
|||
|
|
@ -104,12 +104,12 @@ class S3Cache(BaseCache):
|
|||
Compatible with Python 3.8+.
|
||||
"""
|
||||
try:
|
||||
verbose_logger.debug(f"Set ASYNC S3 Cache: Key={key}. Value={value}")
|
||||
verbose_logger.debug("Set ASYNC S3 Cache: Key=%s. Value=%s", key, value)
|
||||
loop = asyncio.get_event_loop()
|
||||
func = partial(self.set_cache, key, value, **kwargs)
|
||||
await loop.run_in_executor(None, func)
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"S3 Caching: async_set_cache() - Got exception from S3: {e}")
|
||||
verbose_logger.error("S3 Caching: async_set_cache() - Got exception from S3: %s", e)
|
||||
|
||||
def get_cache(self, key, **kwargs):
|
||||
import botocore
|
||||
|
|
@ -138,17 +138,20 @@ class S3Cache(BaseCache):
|
|||
if not isinstance(cached_response, dict):
|
||||
cached_response = dict(cached_response)
|
||||
verbose_logger.debug(
|
||||
f"Got S3 Cache: key: {key}, cached_response {cached_response}. Type Response {type(cached_response)}"
|
||||
"Got S3 Cache: key: %s, cached_response %s. Type Response %s",
|
||||
key,
|
||||
cached_response,
|
||||
type(cached_response),
|
||||
)
|
||||
|
||||
return cached_response
|
||||
except botocore.exceptions.ClientError as e: # type: ignore
|
||||
if e.response["Error"]["Code"] == "NoSuchKey":
|
||||
verbose_logger.debug(f"S3 Cache: The specified key '{key}' does not exist in the S3 bucket.")
|
||||
verbose_logger.debug("S3 Cache: The specified key '%s' does not exist in the S3 bucket.", key)
|
||||
return None
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"S3 Caching: get_cache() - Got exception from S3: {e}")
|
||||
verbose_logger.error("S3 Caching: get_cache() - Got exception from S3: %s", e)
|
||||
|
||||
async def async_get_cache(self, key, **kwargs):
|
||||
"""
|
||||
|
|
@ -156,13 +159,13 @@ class S3Cache(BaseCache):
|
|||
Compatible with Python 3.8+.
|
||||
"""
|
||||
try:
|
||||
verbose_logger.debug(f"Get ASYNC S3 Cache: key: {key}")
|
||||
verbose_logger.debug("Get ASYNC S3 Cache: key: %s", key)
|
||||
loop = asyncio.get_event_loop()
|
||||
func = partial(self.get_cache, key, **kwargs)
|
||||
result = await loop.run_in_executor(None, func)
|
||||
return result
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"S3 Caching: async_get_cache() - Got exception from S3: {e}")
|
||||
verbose_logger.error("S3 Caching: async_get_cache() - Got exception from S3: %s", e)
|
||||
return None
|
||||
|
||||
def flush_cache(self):
|
||||
|
|
|
|||
|
|
@ -249,7 +249,7 @@ class ValkeySemanticCache(RedisSemanticCache):
|
|||
if ttl is not None:
|
||||
self.sync_client.expire(doc_key, ttl)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error in Valkey semantic-cache set_cache: {e!s}")
|
||||
print_verbose(f"Error in Valkey semantic-cache set_cache: {e}")
|
||||
|
||||
def get_cache(self, key: str, **kwargs: Any) -> Any:
|
||||
print_verbose(f"Valkey semantic-cache get_cache, kwargs: {kwargs}")
|
||||
|
|
@ -268,7 +268,7 @@ class ValkeySemanticCache(RedisSemanticCache):
|
|||
)
|
||||
return self._resolve_hit(self._first_hit(search_result), key, **kwargs)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error in Valkey semantic-cache get_cache: {e!s}")
|
||||
print_verbose(f"Error in Valkey semantic-cache get_cache: {e}")
|
||||
kwargs.setdefault("metadata", {})["semantic-similarity"] = 0.0
|
||||
|
||||
async def async_set_cache(self, key: str, value: Any, **kwargs: Any) -> None:
|
||||
|
|
@ -288,7 +288,7 @@ class ValkeySemanticCache(RedisSemanticCache):
|
|||
if ttl is not None:
|
||||
await self.async_client.expire(doc_key, ttl)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error in async Valkey semantic-cache set_cache: {e!s}")
|
||||
print_verbose(f"Error in async Valkey semantic-cache set_cache: {e}")
|
||||
|
||||
async def async_get_cache(self, key: str, **kwargs: Any) -> Any:
|
||||
print_verbose(f"Async Valkey semantic-cache get_cache, kwargs: {kwargs}")
|
||||
|
|
@ -307,14 +307,14 @@ class ValkeySemanticCache(RedisSemanticCache):
|
|||
)
|
||||
return self._resolve_hit(self._first_hit(search_result), key, **kwargs)
|
||||
except Exception as e:
|
||||
print_verbose(f"Error in async Valkey semantic-cache get_cache: {e!s}")
|
||||
print_verbose(f"Error in async Valkey semantic-cache get_cache: {e}")
|
||||
kwargs.setdefault("metadata", {})["semantic-similarity"] = 0.0
|
||||
|
||||
async def async_set_cache_pipeline(self, cache_list: list[tuple[str, Any]], **kwargs: Any) -> None:
|
||||
try:
|
||||
await asyncio.gather(*[self.async_set_cache(key, value, **kwargs) for key, value in cache_list])
|
||||
except Exception as e:
|
||||
print_verbose(f"Error in Valkey semantic-cache async_set_cache_pipeline: {e!s}")
|
||||
print_verbose(f"Error in Valkey semantic-cache async_set_cache_pipeline: {e}")
|
||||
|
||||
async def _index_info(self) -> dict:
|
||||
return await self.async_client.ft(self.index_name).info()
|
||||
|
|
|
|||
|
|
@ -465,7 +465,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
self._map_optional_params_to_responses_api_request(optional_params, responses_api_request)
|
||||
|
||||
stream = optional_params.get("stream") or litellm_params.get("stream", False)
|
||||
verbose_logger.debug(f"Chat provider: Stream parameter: {stream}")
|
||||
verbose_logger.debug("Chat provider: Stream parameter: %s", stream)
|
||||
|
||||
# Ensure stream is properly set in the request
|
||||
if stream:
|
||||
|
|
@ -475,7 +475,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
previous_response_id = optional_params.get("previous_response_id")
|
||||
if previous_response_id:
|
||||
# Use the existing session handler for responses API
|
||||
verbose_logger.debug(f"Chat provider: Warning ignoring previous response ID: {previous_response_id}")
|
||||
verbose_logger.debug("Chat provider: Warning ignoring previous response ID: %s", previous_response_id)
|
||||
|
||||
# Convert back to responses API format for the actual request
|
||||
|
||||
|
|
@ -495,7 +495,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
"client": client,
|
||||
}
|
||||
|
||||
verbose_logger.debug(f"Chat provider: Final request model={api_model}, input_items={len(input_items)}")
|
||||
verbose_logger.debug("Chat provider: Final request model=%s, input_items=%s", api_model, len(input_items))
|
||||
|
||||
self._merge_responses_api_request_into_request_data(request_data, responses_api_request, instructions)
|
||||
|
||||
|
|
@ -843,29 +843,29 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
"""Convert chat completion content to responses API format"""
|
||||
from litellm.types.llms.openai import ChatCompletionImageObject
|
||||
|
||||
verbose_logger.debug(f"Chat provider: Converting content to responses format - input type: {type(content)}")
|
||||
verbose_logger.debug("Chat provider: Converting content to responses format - input type: %s", type(content))
|
||||
|
||||
if content is None:
|
||||
return [self._convert_content_str_to_input_text("", role)]
|
||||
elif isinstance(content, str):
|
||||
result = [self._convert_content_str_to_input_text(content, role)]
|
||||
verbose_logger.debug(f"Chat provider: String content -> {result}")
|
||||
verbose_logger.debug("Chat provider: String content -> %s", result)
|
||||
return result
|
||||
elif isinstance(content, list):
|
||||
result = []
|
||||
for i, item in enumerate(content):
|
||||
verbose_logger.debug(f"Chat provider: Processing content item {i}: {type(item)} = {item}")
|
||||
verbose_logger.debug("Chat provider: Processing content item %s: %s = %s", i, type(item), item)
|
||||
if isinstance(item, str):
|
||||
converted = self._convert_content_str_to_input_text(item, role)
|
||||
result.append(converted)
|
||||
verbose_logger.debug(f"Chat provider: -> {converted}")
|
||||
verbose_logger.debug("Chat provider: -> %s", converted)
|
||||
elif isinstance(item, dict):
|
||||
# Handle multimodal content
|
||||
original_type = item.get("type")
|
||||
if original_type == "text":
|
||||
converted = self._convert_content_str_to_input_text(item.get("text", ""), role)
|
||||
result.append(converted)
|
||||
verbose_logger.debug(f"Chat provider: text -> {converted}")
|
||||
verbose_logger.debug("Chat provider: text -> %s", converted)
|
||||
elif original_type == "image_url":
|
||||
# Map to responses API image format
|
||||
converted = cast(
|
||||
|
|
@ -875,14 +875,14 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
),
|
||||
)
|
||||
result.append(converted)
|
||||
verbose_logger.debug(f"Chat provider: image_url -> {converted}")
|
||||
verbose_logger.debug("Chat provider: image_url -> %s", converted)
|
||||
else:
|
||||
# Try to map other types to responses API format
|
||||
item_type = original_type or "input_text"
|
||||
if item_type == "image":
|
||||
converted = {"type": "input_image", **item}
|
||||
result.append(converted)
|
||||
verbose_logger.debug(f"Chat provider: image -> {converted}")
|
||||
verbose_logger.debug("Chat provider: image -> %s", converted)
|
||||
elif item_type == "file":
|
||||
# Map Chat Completion file to Responses API input_file
|
||||
# {"type": "file", "file": {"file_data": "...", "filename": "..."}}
|
||||
|
|
@ -894,7 +894,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
if key in file_data:
|
||||
converted[key] = file_data[key]
|
||||
result.append(converted)
|
||||
verbose_logger.debug(f"Chat provider: file -> {converted}")
|
||||
verbose_logger.debug("Chat provider: file -> %s", converted)
|
||||
elif item_type in [
|
||||
"input_text",
|
||||
"input_image",
|
||||
|
|
@ -906,17 +906,17 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
]:
|
||||
# Already in responses API format
|
||||
result.append(item)
|
||||
verbose_logger.debug(f"Chat provider: passthrough -> {item}")
|
||||
verbose_logger.debug("Chat provider: passthrough -> %s", item)
|
||||
else:
|
||||
# Default to input_text for unknown types
|
||||
converted = self._convert_content_str_to_input_text(str(item.get("text", item)), role)
|
||||
result.append(converted)
|
||||
verbose_logger.debug(f"Chat provider: unknown({original_type}) -> {converted}")
|
||||
verbose_logger.debug(f"Chat provider: Final converted content: {result}")
|
||||
verbose_logger.debug("Chat provider: unknown(%s) -> %s", original_type, converted)
|
||||
verbose_logger.debug("Chat provider: Final converted content: %s", result)
|
||||
return result
|
||||
else:
|
||||
result = [self._convert_content_str_to_input_text(str(content), role)]
|
||||
verbose_logger.debug(f"Chat provider: Other content type -> {result}")
|
||||
verbose_logger.debug("Chat provider: Other content type -> %s", result)
|
||||
return result
|
||||
|
||||
def _convert_tools_to_responses_format(self, tools: list[dict[str, Any]]) -> list["ALL_RESPONSES_API_TOOL_PARAMS"]:
|
||||
|
|
@ -1111,13 +1111,13 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
annotation_dict = annotation
|
||||
else:
|
||||
# Skip unsupported annotation types
|
||||
verbose_logger.debug(f"Skipping unsupported annotation type: {type(annotation)}")
|
||||
verbose_logger.debug("Skipping unsupported annotation type: %s", type(annotation))
|
||||
continue
|
||||
|
||||
result.append(annotation_dict) # type: ignore
|
||||
except Exception as e:
|
||||
# Skip malformed annotations
|
||||
verbose_logger.debug(f"Skipping malformed annotation: {annotation}, error: {e}")
|
||||
verbose_logger.debug("Skipping malformed annotation: %s, error: %s", annotation, e)
|
||||
continue
|
||||
|
||||
return result if result else None
|
||||
|
|
@ -1222,11 +1222,11 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
):
|
||||
return ModelResponseStream(**parsed_chunk)
|
||||
|
||||
verbose_logger.debug(f"Chat provider: Processing event type: {event_type}")
|
||||
verbose_logger.debug("Chat provider: Processing event type: %s", event_type)
|
||||
|
||||
if event_type == "response.created":
|
||||
# Initial response creation event
|
||||
verbose_logger.debug(f"Chat provider: response.created -> {parsed_chunk}")
|
||||
verbose_logger.debug("Chat provider: response.created -> %s", parsed_chunk)
|
||||
return ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
|
|
@ -1434,7 +1434,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
else:
|
||||
pass
|
||||
# For any unhandled event types, create a minimal valid chunk or skip
|
||||
verbose_logger.debug(f"Chat provider: Unhandled event type '{event_type}', creating empty chunk")
|
||||
verbose_logger.debug("Chat provider: Unhandled event type '%s', creating empty chunk", event_type)
|
||||
|
||||
# Return a minimal valid chunk for unknown events
|
||||
return ModelResponseStream(
|
||||
|
|
@ -1457,7 +1457,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
Returns:
|
||||
ModelResponseStream: OpenAI-formatted streaming chunk
|
||||
"""
|
||||
verbose_logger.debug(f"Chat provider: transform_streaming_response called with chunk: {chunk}")
|
||||
verbose_logger.debug("Chat provider: transform_streaming_response called with chunk: %s", chunk)
|
||||
return self._with_stream_scoped_id(
|
||||
OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
|
||||
chunk, tool_call_index_map=self._tool_call_index_map
|
||||
|
|
|
|||
|
|
@ -197,6 +197,16 @@ RUNWAYML_POLLING_TIMEOUT = int(os.getenv("RUNWAYML_POLLING_TIMEOUT", 600)) # 10
|
|||
########## Networking constants ##############################################################
|
||||
_DEFAULT_TTL_FOR_HTTPX_CLIENTS = 3600 # 1 hour, re-use the same httpx client for 1 hour
|
||||
|
||||
# The earliest an evicted, litellm-created client may be closed. A request handed the
|
||||
# client just before eviction is still using it, so nothing is closed inside this window;
|
||||
# past it, the client is closed once it reports no connection in flight.
|
||||
EVICTED_LLM_CLIENT_CLOSE_GRACE_SECONDS = 900
|
||||
|
||||
# How many evicted clients may be queued for closing at once. Past this, an evicted client
|
||||
# is left to the collector rather than letting a cache-churning workload grow the queue
|
||||
# without bound. Each queued entry is ~100 bytes and comes due within one grace window.
|
||||
EVICTED_LLM_CLIENT_CLOSE_MAX_PENDING = 10_000
|
||||
|
||||
# Aiohttp connection pooling - prevents memory leaks from unbounded connection growth
|
||||
# Set to 0 for unlimited (not recommended for production)
|
||||
AIOHTTP_CONNECTOR_LIMIT = int(os.getenv("AIOHTTP_CONNECTOR_LIMIT", 1000))
|
||||
|
|
@ -303,6 +313,7 @@ MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES = int(os.getenv("MAX_LONG_SIDE_FOR_IMAGE_HIGH_R
|
|||
MAX_TILE_WIDTH = int(os.getenv("MAX_TILE_WIDTH", 512))
|
||||
MAX_TILE_HEIGHT = int(os.getenv("MAX_TILE_HEIGHT", 512))
|
||||
OPENAI_FILE_SEARCH_COST_PER_1K_CALLS = float(os.getenv("OPENAI_FILE_SEARCH_COST_PER_1K_CALLS", 2.5 / 1000))
|
||||
GROQ_BROWSER_VISIT_WEBSITE_COST_PER_CALL = 1.0 / 1000
|
||||
# Azure OpenAI Assistants feature costs
|
||||
# Source: https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/
|
||||
AZURE_FILE_SEARCH_COST_PER_GB_PER_DAY = float(
|
||||
|
|
|
|||
|
|
@ -273,7 +273,7 @@ def _get_additional_costs(
|
|||
completion_tokens=completion_tokens,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Error calculating additional costs: {e}")
|
||||
verbose_logger.debug("Error calculating additional costs: %s", e)
|
||||
|
||||
return None
|
||||
|
||||
|
|
@ -715,7 +715,7 @@ def _get_provider_for_cost_calc(
|
|||
_, custom_llm_provider, _, _ = litellm.get_llm_provider(model=model)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(
|
||||
f"litellm.cost_calculator.py::_get_provider_for_cost_calc() - Error inferring custom_llm_provider - {e!s}"
|
||||
"litellm.cost_calculator.py::_get_provider_for_cost_calc() - Error inferring custom_llm_provider - %s", e
|
||||
)
|
||||
return None
|
||||
|
||||
|
|
@ -896,7 +896,7 @@ def _get_usage_object(
|
|||
elif isinstance(usage_obj, BaseModel):
|
||||
return Usage(**usage_obj.model_dump())
|
||||
else:
|
||||
verbose_logger.debug(f"Unknown usage object type: {type(usage_obj)}, usage_obj: {usage_obj}")
|
||||
verbose_logger.debug("Unknown usage object type: %s, usage_obj: %s", type(usage_obj), usage_obj)
|
||||
return None
|
||||
|
||||
|
||||
|
|
@ -994,16 +994,17 @@ def _apply_cost_margin(
|
|||
if custom_llm_provider and custom_llm_provider in litellm.cost_margin_config:
|
||||
margin_config = litellm.cost_margin_config[custom_llm_provider]
|
||||
if verbose_logger.isEnabledFor(logging.DEBUG):
|
||||
verbose_logger.debug(f"Found provider-specific margin config for {custom_llm_provider}: {margin_config}")
|
||||
verbose_logger.debug("Found provider-specific margin config for %s: %s", custom_llm_provider, margin_config)
|
||||
elif "global" in litellm.cost_margin_config:
|
||||
margin_config = litellm.cost_margin_config["global"]
|
||||
if verbose_logger.isEnabledFor(logging.DEBUG):
|
||||
verbose_logger.debug(f"Using global margin config: {margin_config}")
|
||||
verbose_logger.debug("Using global margin config: %s", margin_config)
|
||||
else:
|
||||
if verbose_logger.isEnabledFor(logging.DEBUG):
|
||||
verbose_logger.debug(
|
||||
f"No margin config found. Provider: {custom_llm_provider}, "
|
||||
f"Available configs: {list(litellm.cost_margin_config.keys())}"
|
||||
"No margin config found. Provider: %s, Available configs: %s",
|
||||
custom_llm_provider,
|
||||
list(litellm.cost_margin_config.keys()),
|
||||
)
|
||||
|
||||
if margin_config is not None:
|
||||
|
|
@ -1051,6 +1052,8 @@ def _store_cost_breakdown_in_logging_obj(
|
|||
cache_read_cost: float | None = None,
|
||||
cache_creation_cost: float | None = None,
|
||||
reasoning_cost: float | None = None,
|
||||
service_tier: str | None = None,
|
||||
data_residency: str | None = None,
|
||||
) -> None:
|
||||
"""
|
||||
Helper function to store cost breakdown in the logging object.
|
||||
|
|
@ -1068,6 +1071,8 @@ def _store_cost_breakdown_in_logging_obj(
|
|||
margin_percent: Margin percentage applied (0.10 = 10%)
|
||||
margin_fixed_amount: Fixed margin amount in USD
|
||||
margin_total_amount: Total margin added in USD
|
||||
service_tier: Tier the costs above were priced on, already resolved
|
||||
data_residency: Region uplift the costs above were priced on, already resolved
|
||||
"""
|
||||
if litellm_logging_obj is None:
|
||||
return
|
||||
|
|
@ -1089,10 +1094,12 @@ def _store_cost_breakdown_in_logging_obj(
|
|||
cache_read_cost=cache_read_cost,
|
||||
cache_creation_cost=cache_creation_cost,
|
||||
reasoning_cost=reasoning_cost,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
|
||||
except Exception as breakdown_error:
|
||||
verbose_logger.debug(f"Error storing cost breakdown: {breakdown_error!s}")
|
||||
verbose_logger.debug("Error storing cost breakdown: %s", breakdown_error)
|
||||
# Don't fail the main cost calculation if breakdown storage fails
|
||||
|
||||
|
||||
|
|
@ -1219,7 +1226,7 @@ def completion_cost(
|
|||
for idx, model in enumerate(potential_model_names):
|
||||
try:
|
||||
if verbose_logger.isEnabledFor(logging.DEBUG):
|
||||
verbose_logger.debug(f"selected model name for cost calculation: {model}")
|
||||
verbose_logger.debug("selected model name for cost calculation: %s", model)
|
||||
|
||||
if completion_response is not None and (
|
||||
isinstance(completion_response, BaseModel) or isinstance(completion_response, dict)
|
||||
|
|
@ -1315,7 +1322,8 @@ def completion_cost(
|
|||
) # strip the llm provider from the model name -> for image gen cost calculation
|
||||
except Exception as e:
|
||||
verbose_logger.debug(
|
||||
f"litellm.cost_calculator.py::completion_cost() - Error inferring custom_llm_provider - {e!s}"
|
||||
"litellm.cost_calculator.py::completion_cost() - Error inferring custom_llm_provider - %s",
|
||||
e,
|
||||
)
|
||||
if CostCalculatorUtils._call_type_has_image_response(call_type) and isinstance(
|
||||
completion_response, ImageResponse
|
||||
|
|
@ -1469,6 +1477,8 @@ def completion_cost(
|
|||
margin_percent=margin_percent,
|
||||
margin_fixed_amount=margin_fixed_amount,
|
||||
margin_total_amount=margin_total_amount,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
|
||||
return _final_cost
|
||||
|
|
@ -1657,12 +1667,14 @@ def completion_cost(
|
|||
cache_read_cost=_cache_read_cost,
|
||||
cache_creation_cost=_cache_creation_cost,
|
||||
reasoning_cost=_reasoning_cost,
|
||||
service_tier=service_tier,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
|
||||
return _final_cost
|
||||
except Exception as e:
|
||||
verbose_logger.debug(
|
||||
f"litellm.cost_calculator.py::completion_cost() - Error calculating cost for model={model} - {e!s}"
|
||||
"litellm.cost_calculator.py::completion_cost() - Error calculating cost for model=%s - %s", model, e
|
||||
)
|
||||
if idx == len(potential_model_names) - 1:
|
||||
raise e
|
||||
|
|
@ -1878,7 +1890,7 @@ def vector_store_search_cost(
|
|||
)
|
||||
|
||||
if config is None:
|
||||
verbose_logger.debug(f"Vector store search is not supported for {custom_llm_provider}")
|
||||
verbose_logger.debug("Vector store search is not supported for %s", custom_llm_provider)
|
||||
return 0.0, 0.0
|
||||
|
||||
return config.calculate_vector_store_cost(
|
||||
|
|
@ -1966,7 +1978,7 @@ def default_image_cost_calculator(
|
|||
# gpt-image-1 models use low, medium, high quality. If user did not specify quality, use medium fot gpt-image-1 model family
|
||||
model_name_with_v2_quality = f"{ImageGenerationRequestQuality.HIGH.value}/{base_model_name}"
|
||||
|
||||
verbose_logger.debug(f"Looking up cost for models: {model_name_with_quality}, {base_model_name}")
|
||||
verbose_logger.debug("Looking up cost for models: %s, %s", model_name_with_quality, base_model_name)
|
||||
|
||||
model_without_provider = f"{size_str}/{model.split('/')[-1]}"
|
||||
model_with_quality_without_provider = f"{quality}/{model_without_provider}" if quality else model_without_provider
|
||||
|
|
@ -2036,7 +2048,7 @@ def default_video_cost_calculator(
|
|||
model_name_without_custom_llm_provider = model.replace(f"{custom_llm_provider}/", "")
|
||||
base_model_name = f"{custom_llm_provider}/{model_name_without_custom_llm_provider}"
|
||||
|
||||
verbose_logger.debug(f"Looking up cost for video model: {base_model_name}")
|
||||
verbose_logger.debug("Looking up cost for video model: %s", base_model_name)
|
||||
|
||||
model_without_provider = model.split("/")[-1]
|
||||
|
||||
|
|
@ -2072,7 +2084,8 @@ def default_video_cost_calculator(
|
|||
|
||||
# If no cost information found, return 0
|
||||
verbose_logger.info(
|
||||
f"No cost information found for video model {model}. Please add pricing to model_prices_and_context_window.json"
|
||||
"No cost information found for video model %s. Please add pricing to model_prices_and_context_window.json",
|
||||
model,
|
||||
)
|
||||
return 0.0
|
||||
|
||||
|
|
@ -2351,6 +2364,7 @@ def handle_realtime_stream_cost_calculation(
|
|||
cost_for_built_in_tools_cost_usd_dollar=0.0,
|
||||
total_cost_usd_dollar=total_cost,
|
||||
additional_costs={"transcription_cost": transcription_cost} if transcription_cost > 0 else None,
|
||||
data_residency=data_residency,
|
||||
)
|
||||
|
||||
return total_cost
|
||||
|
|
|
|||
|
|
@ -1140,7 +1140,7 @@ class MidStreamFallbackError(ServiceUnavailableError): # type: ignore
|
|||
if self.max_retries:
|
||||
_message += f", LiteLLM Max Retries: {self.max_retries}"
|
||||
if self.original_exception:
|
||||
_message += f" Original exception: {type(self.original_exception).__name__}: {self.original_exception!s}"
|
||||
_message += f" Original exception: {type(self.original_exception).__name__}: {self.original_exception}"
|
||||
return _message
|
||||
|
||||
def __repr__(self):
|
||||
|
|
|
|||
|
|
@ -364,7 +364,7 @@ class MCPClient:
|
|||
try:
|
||||
await session_ctx.__aexit__(None, None, None)
|
||||
except BaseException as e:
|
||||
verbose_logger.debug(f"Error during session context exit: {e}")
|
||||
verbose_logger.debug("Error during session context exit: %s", e)
|
||||
except BaseException as e:
|
||||
in_flight_error = e
|
||||
raise
|
||||
|
|
@ -372,7 +372,7 @@ class MCPClient:
|
|||
try:
|
||||
await transport_ctx.__aexit__(None, None, None)
|
||||
except BaseException as exit_error:
|
||||
verbose_logger.debug(f"Error during transport context exit: {exit_error}")
|
||||
verbose_logger.debug("Error during transport context exit: %s", exit_error)
|
||||
root_cause = _first_non_cancelled_cause(exit_error)
|
||||
if root_cause is not None and isinstance(in_flight_error, asyncio.CancelledError):
|
||||
raise root_cause from in_flight_error
|
||||
|
|
@ -402,7 +402,7 @@ class MCPClient:
|
|||
try:
|
||||
await http_client.aclose()
|
||||
except BaseException as e:
|
||||
verbose_logger.debug(f"Error during http_client cleanup: {e}")
|
||||
verbose_logger.debug("Error during http_client cleanup: %s", e)
|
||||
|
||||
def update_auth_value(self, mcp_auth_value: str | dict[str, str]):
|
||||
"""
|
||||
|
|
@ -464,7 +464,7 @@ class MCPClient:
|
|||
"""Create an httpx.AsyncClient with LiteLLM's SSL configuration."""
|
||||
# Get unified SSL configuration using the same logic as http_handler.py
|
||||
ssl_config = get_ssl_configuration(self.ssl_verify)
|
||||
verbose_logger.debug(f"MCP client using SSL configuration: {type(ssl_config).__name__}")
|
||||
verbose_logger.debug("MCP client using SSL configuration: %s", type(ssl_config).__name__)
|
||||
# The MCP SDK's sse_client and streamable_http_client call this factory without
|
||||
# passing auth=, so the fallback is used: a v2-resolved auth if present, else the
|
||||
# SigV4 aws_auth. Both are None for the common case — no behavior change.
|
||||
|
|
@ -490,7 +490,7 @@ class MCPClient:
|
|||
MCP client (triggering the upstream OAuth flow) rather than
|
||||
masking them as "connected, no tools".
|
||||
"""
|
||||
verbose_logger.debug(f"MCP client listing tools from {self.server_url or 'stdio'}")
|
||||
verbose_logger.debug("MCP client listing tools from %s", self.server_url or "stdio")
|
||||
|
||||
async def _list_tools_operation(session: ClientSession):
|
||||
return await session.list_tools()
|
||||
|
|
@ -499,7 +499,9 @@ class MCPClient:
|
|||
result = await self.run_with_session(_list_tools_operation, quiet_on_error=raise_on_error)
|
||||
tool_count = len(result.tools)
|
||||
tool_names = [tool.name for tool in result.tools]
|
||||
verbose_logger.info(f"MCP client listed {tool_count} tools from {self.server_url or 'stdio'}: {tool_names}")
|
||||
verbose_logger.info(
|
||||
"MCP client listed %s tools from %s: %s", tool_count, self.server_url or "stdio", tool_names
|
||||
)
|
||||
return result.tools
|
||||
except asyncio.CancelledError:
|
||||
verbose_logger.warning("MCP client list_tools was cancelled")
|
||||
|
|
@ -515,7 +517,7 @@ class MCPClient:
|
|||
_log(
|
||||
f"MCP client list_tools failed - "
|
||||
f"Error Type: {error_type}, "
|
||||
f"Error: {e!s}, "
|
||||
f"Error: {e}, "
|
||||
f"Server: {self.server_url or 'stdio'}, "
|
||||
f"Transport: {self.transport_type}"
|
||||
)
|
||||
|
|
@ -536,7 +538,7 @@ class MCPClient:
|
|||
def error_tool_result(exc: Exception) -> MCPCallToolResult:
|
||||
"""The error result ``call_tool`` returns when it swallows a failure (no re-execution)."""
|
||||
return MCPCallToolResult(
|
||||
content=[TextContent(type="text", text=f"{type(exc).__name__}: {exc!s}")],
|
||||
content=[TextContent(type="text", text=f"{type(exc).__name__}: {exc}")],
|
||||
isError=True,
|
||||
)
|
||||
|
||||
|
|
@ -555,7 +557,7 @@ class MCPClient:
|
|||
an upstream 401 so it can re-mint the exchanged token and retry once; every other
|
||||
caller keeps the default and gets graceful ``isError`` degradation.
|
||||
"""
|
||||
verbose_logger.info(f"MCP client calling tool '{call_tool_request_params.name}'")
|
||||
verbose_logger.info("MCP client calling tool '%s'", call_tool_request_params.name)
|
||||
|
||||
async def on_progress(progress: float, total: float | None, message: str | None):
|
||||
percentage = (progress / total * 100) if total else 0
|
||||
|
|
@ -568,7 +570,7 @@ class MCPClient:
|
|||
try:
|
||||
await host_progress_callback(progress, total)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Failed to forward to Host: {e}")
|
||||
verbose_logger.warning("Failed to forward to Host: %s", e)
|
||||
|
||||
async def _call_tool_operation(session: ClientSession):
|
||||
verbose_logger.debug("MCP client sending tool call to session")
|
||||
|
|
@ -580,16 +582,16 @@ class MCPClient:
|
|||
|
||||
try:
|
||||
tool_result = await self.run_with_session(_call_tool_operation, quiet_on_error=raise_on_error)
|
||||
verbose_logger.info(f"MCP client tool call '{call_tool_request_params.name}' completed successfully")
|
||||
verbose_logger.info("MCP client tool call '%s' completed successfully", call_tool_request_params.name)
|
||||
return tool_result
|
||||
except asyncio.CancelledError:
|
||||
verbose_logger.warning(f"MCP client tool call timed out after {self.timeout}s for {self.server_url}")
|
||||
verbose_logger.warning("MCP client tool call timed out after %ss for %s", self.timeout, self.server_url)
|
||||
raise
|
||||
except Exception as e:
|
||||
import traceback
|
||||
|
||||
error_trace = traceback.format_exc()
|
||||
verbose_logger.debug(f"MCP client tool call traceback:\n{error_trace}")
|
||||
verbose_logger.debug("MCP client tool call traceback:\n%s", error_trace)
|
||||
# Log detailed error information
|
||||
error_type = type(e).__name__
|
||||
# When the caller opted into raise_on_error it owns the exception and logs it at the
|
||||
|
|
@ -601,7 +603,7 @@ class MCPClient:
|
|||
_log(
|
||||
f"MCP client call_tool failed - "
|
||||
f"Error Type: {error_type}, "
|
||||
f"Error: {e!s}, "
|
||||
f"Error: {e}, "
|
||||
f"Tool: {call_tool_request_params.name}, "
|
||||
f"Server: {self.server_url or 'stdio'}, "
|
||||
f"Transport: {self.transport_type}"
|
||||
|
|
@ -619,7 +621,7 @@ class MCPClient:
|
|||
|
||||
async def list_prompts(self) -> list[Prompt]:
|
||||
"""List available prompts from the server."""
|
||||
verbose_logger.debug(f"MCP client listing tools from {self.server_url or 'stdio'}")
|
||||
verbose_logger.debug("MCP client listing tools from %s", self.server_url or "stdio")
|
||||
|
||||
async def _list_prompts_operation(session: ClientSession):
|
||||
return await session.list_prompts()
|
||||
|
|
@ -629,7 +631,7 @@ class MCPClient:
|
|||
prompt_count = len(result.prompts)
|
||||
prompt_names = [prompt.name for prompt in result.prompts]
|
||||
verbose_logger.info(
|
||||
f"MCP client listed {prompt_count} tools from {self.server_url or 'stdio'}: {prompt_names}"
|
||||
"MCP client listed %s tools from %s: %s", prompt_count, self.server_url or "stdio", prompt_names
|
||||
)
|
||||
return result.prompts
|
||||
except asyncio.CancelledError:
|
||||
|
|
@ -638,11 +640,11 @@ class MCPClient:
|
|||
except Exception as e:
|
||||
error_type = type(e).__name__
|
||||
verbose_logger.error(
|
||||
f"MCP client list_prompts failed - "
|
||||
f"Error Type: {error_type}, "
|
||||
f"Error: {e!s}, "
|
||||
f"Server: {self.server_url or 'stdio'}, "
|
||||
f"Transport: {self.transport_type}"
|
||||
"MCP client list_prompts failed - Error Type: %s, Error: %s, Server: %s, Transport: %s",
|
||||
error_type,
|
||||
e,
|
||||
self.server_url or "stdio",
|
||||
self.transport_type,
|
||||
)
|
||||
# Check if it's a stream/connection error
|
||||
if "BrokenResourceError" in error_type or "Broken" in error_type:
|
||||
|
|
@ -655,7 +657,7 @@ class MCPClient:
|
|||
|
||||
async def get_prompt(self, get_prompt_request_params: GetPromptRequestParams) -> GetPromptResult:
|
||||
"""Fetch a prompt definition from the MCP server."""
|
||||
verbose_logger.info(f"MCP client fetching prompt '{get_prompt_request_params.name}'")
|
||||
verbose_logger.info("MCP client fetching prompt '%s'", get_prompt_request_params.name)
|
||||
|
||||
async def _get_prompt_operation(session: ClientSession):
|
||||
verbose_logger.debug("MCP client sending get_prompt request to session")
|
||||
|
|
@ -666,7 +668,7 @@ class MCPClient:
|
|||
|
||||
try:
|
||||
get_prompt_result = await self.run_with_session(_get_prompt_operation)
|
||||
verbose_logger.info(f"MCP client get_prompt '{get_prompt_request_params.name}' completed successfully")
|
||||
verbose_logger.info("MCP client get_prompt '%s' completed successfully", get_prompt_request_params.name)
|
||||
return get_prompt_result
|
||||
except asyncio.CancelledError:
|
||||
verbose_logger.warning("MCP client get_prompt was cancelled")
|
||||
|
|
@ -675,16 +677,16 @@ class MCPClient:
|
|||
import traceback
|
||||
|
||||
error_trace = traceback.format_exc()
|
||||
verbose_logger.debug(f"MCP client get_prompt traceback:\n{error_trace}")
|
||||
verbose_logger.debug("MCP client get_prompt traceback:\n%s", error_trace)
|
||||
# Log detailed error information
|
||||
error_type = type(e).__name__
|
||||
verbose_logger.error(
|
||||
f"MCP client get_prompt failed - "
|
||||
f"Error Type: {error_type}, "
|
||||
f"Error: {e!s}, "
|
||||
f"Prompt: {get_prompt_request_params.name}, "
|
||||
f"Server: {self.server_url or 'stdio'}, "
|
||||
f"Transport: {self.transport_type}"
|
||||
"MCP client get_prompt failed - Error Type: %s, Error: %s, Prompt: %s, Server: %s, Transport: %s",
|
||||
error_type,
|
||||
e,
|
||||
get_prompt_request_params.name,
|
||||
self.server_url or "stdio",
|
||||
self.transport_type,
|
||||
)
|
||||
# Check if it's a stream/connection error
|
||||
if "BrokenResourceError" in error_type or "Broken" in error_type:
|
||||
|
|
@ -696,7 +698,7 @@ class MCPClient:
|
|||
|
||||
async def list_resources(self) -> list[Resource]:
|
||||
"""List available resources from the server."""
|
||||
verbose_logger.debug(f"MCP client listing resources from {self.server_url or 'stdio'}")
|
||||
verbose_logger.debug("MCP client listing resources from %s", self.server_url or "stdio")
|
||||
|
||||
async def _list_resources_operation(session: ClientSession):
|
||||
return await session.list_resources()
|
||||
|
|
@ -706,7 +708,7 @@ class MCPClient:
|
|||
resource_count = len(result.resources)
|
||||
resource_names = [resource.name for resource in result.resources]
|
||||
verbose_logger.info(
|
||||
f"MCP client listed {resource_count} resources from {self.server_url or 'stdio'}: {resource_names}"
|
||||
"MCP client listed %s resources from %s: %s", resource_count, self.server_url or "stdio", resource_names
|
||||
)
|
||||
return result.resources
|
||||
except asyncio.CancelledError:
|
||||
|
|
@ -715,11 +717,11 @@ class MCPClient:
|
|||
except Exception as e:
|
||||
error_type = type(e).__name__
|
||||
verbose_logger.error(
|
||||
f"MCP client list_resources failed - "
|
||||
f"Error Type: {error_type}, "
|
||||
f"Error: {e!s}, "
|
||||
f"Server: {self.server_url or 'stdio'}, "
|
||||
f"Transport: {self.transport_type}"
|
||||
"MCP client list_resources failed - Error Type: %s, Error: %s, Server: %s, Transport: %s",
|
||||
error_type,
|
||||
e,
|
||||
self.server_url or "stdio",
|
||||
self.transport_type,
|
||||
)
|
||||
# Check if it's a stream/connection error
|
||||
if "BrokenResourceError" in error_type or "Broken" in error_type:
|
||||
|
|
@ -732,7 +734,7 @@ class MCPClient:
|
|||
|
||||
async def list_resource_templates(self) -> list[ResourceTemplate]:
|
||||
"""List available resource templates from the server."""
|
||||
verbose_logger.debug(f"MCP client listing resource templates from {self.server_url or 'stdio'}")
|
||||
verbose_logger.debug("MCP client listing resource templates from %s", self.server_url or "stdio")
|
||||
|
||||
async def _list_resource_templates_operation(session: ClientSession):
|
||||
return await session.list_resource_templates()
|
||||
|
|
@ -742,7 +744,10 @@ class MCPClient:
|
|||
resource_template_count = len(result.resourceTemplates)
|
||||
resource_template_names = [resourceTemplate.name for resourceTemplate in result.resourceTemplates]
|
||||
verbose_logger.info(
|
||||
f"MCP client listed {resource_template_count} resource templates from {self.server_url or 'stdio'}: {resource_template_names}"
|
||||
"MCP client listed %s resource templates from %s: %s",
|
||||
resource_template_count,
|
||||
self.server_url or "stdio",
|
||||
resource_template_names,
|
||||
)
|
||||
return result.resourceTemplates
|
||||
except asyncio.CancelledError:
|
||||
|
|
@ -751,11 +756,11 @@ class MCPClient:
|
|||
except Exception as e:
|
||||
error_type = type(e).__name__
|
||||
verbose_logger.error(
|
||||
f"MCP client list_resource_templates failed - "
|
||||
f"Error Type: {error_type}, "
|
||||
f"Error: {e!s}, "
|
||||
f"Server: {self.server_url or 'stdio'}, "
|
||||
f"Transport: {self.transport_type}"
|
||||
"MCP client list_resource_templates failed - Error Type: %s, Error: %s, Server: %s, Transport: %s",
|
||||
error_type,
|
||||
e,
|
||||
self.server_url or "stdio",
|
||||
self.transport_type,
|
||||
)
|
||||
# Check if it's a stream/connection error
|
||||
if "BrokenResourceError" in error_type or "Broken" in error_type:
|
||||
|
|
@ -768,7 +773,7 @@ class MCPClient:
|
|||
|
||||
async def read_resource(self, url: AnyUrl) -> ReadResourceResult:
|
||||
"""Fetch resource contents from the MCP server."""
|
||||
verbose_logger.info(f"MCP client fetching resource '{url}'")
|
||||
verbose_logger.info("MCP client fetching resource '%s'", url)
|
||||
|
||||
async def _read_resource_operation(session: ClientSession):
|
||||
verbose_logger.debug("MCP client sending read_resource request to session")
|
||||
|
|
@ -776,7 +781,7 @@ class MCPClient:
|
|||
|
||||
try:
|
||||
read_resource_result = await self.run_with_session(_read_resource_operation)
|
||||
verbose_logger.info(f"MCP client read_resource '{url}' completed successfully")
|
||||
verbose_logger.info("MCP client read_resource '%s' completed successfully", url)
|
||||
return read_resource_result
|
||||
except asyncio.CancelledError:
|
||||
verbose_logger.warning("MCP client read_resource was cancelled")
|
||||
|
|
@ -785,16 +790,16 @@ class MCPClient:
|
|||
import traceback
|
||||
|
||||
error_trace = traceback.format_exc()
|
||||
verbose_logger.debug(f"MCP client read_resource traceback:\n{error_trace}")
|
||||
verbose_logger.debug("MCP client read_resource traceback:\n%s", error_trace)
|
||||
# Log detailed error information
|
||||
error_type = type(e).__name__
|
||||
verbose_logger.error(
|
||||
f"MCP client read_resource failed - "
|
||||
f"Error Type: {error_type}, "
|
||||
f"Error: {e!s}, "
|
||||
f"Url: {url}, "
|
||||
f"Server: {self.server_url or 'stdio'}, "
|
||||
f"Transport: {self.transport_type}"
|
||||
"MCP client read_resource failed - Error Type: %s, Error: %s, Url: %s, Server: %s, Transport: %s",
|
||||
error_type,
|
||||
e,
|
||||
url,
|
||||
self.server_url or "stdio",
|
||||
self.transport_type,
|
||||
)
|
||||
# Check if it's a stream/connection error
|
||||
if "BrokenResourceError" in error_type or "Broken" in error_type:
|
||||
|
|
|
|||
|
|
@ -98,7 +98,7 @@ class GenerateContentToCompletionHandler:
|
|||
return generate_content_response
|
||||
|
||||
except Exception as e:
|
||||
raise ValueError(f"Error calling litellm.acompletion for generate_content: {e!s}")
|
||||
raise ValueError(f"Error calling litellm.acompletion for generate_content: {e}")
|
||||
|
||||
@staticmethod
|
||||
def generate_content_handler(
|
||||
|
|
@ -159,4 +159,4 @@ class GenerateContentToCompletionHandler:
|
|||
return generate_content_response
|
||||
|
||||
except Exception as e:
|
||||
raise ValueError(f"Error calling litellm.completion for generate_content: {e!s}")
|
||||
raise ValueError(f"Error calling litellm.completion for generate_content: {e}")
|
||||
|
|
|
|||
|
|
@ -104,9 +104,10 @@ class GoogleGenAIStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
except json.JSONDecodeError:
|
||||
# This can happen if the stream is abruptly cut off mid-argument string.
|
||||
verbose_logger.warning(
|
||||
f"Could not parse tool call arguments at end of stream for index {tool_call_index}. "
|
||||
f"Name: {tool_call_data['name']}. "
|
||||
f"Partial args: {tool_call_data['arguments']}"
|
||||
"Could not parse tool call arguments at end of stream for index %s. Name: %s. Partial args: %s",
|
||||
tool_call_index,
|
||||
tool_call_data["name"],
|
||||
tool_call_data["arguments"],
|
||||
)
|
||||
if parts:
|
||||
final_chunk = {
|
||||
|
|
@ -662,7 +663,7 @@ class GoogleGenAIAdapter:
|
|||
|
||||
# Optimization: Skip chunks that have no new data
|
||||
if not function_name and not args_chunk:
|
||||
verbose_logger.debug(f"Skipping empty tool call chunk for index: {tool_call_index}")
|
||||
verbose_logger.debug("Skipping empty tool call chunk for index: %s", tool_call_index)
|
||||
continue
|
||||
|
||||
if function_name:
|
||||
|
|
|
|||
|
|
@ -68,8 +68,8 @@ async def send_to_webhook(slackAlertingInstance: SlackAlertingType, item, count)
|
|||
data=json.dumps(payload),
|
||||
)
|
||||
if response.status_code != 200:
|
||||
verbose_proxy_logger.debug(f"Error sending slack alert to url={item['url']}. Error={response.text}")
|
||||
verbose_proxy_logger.debug("Error sending slack alert to url=%s. Error=%s", item["url"], response.text)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.debug(f"Error sending slack alert: {e!s}")
|
||||
verbose_proxy_logger.debug("Error sending slack alert: %s", e)
|
||||
finally:
|
||||
_print_alerting_payload_warning(payload, slackAlertingInstance=slackAlertingInstance)
|
||||
|
|
|
|||
|
|
@ -1467,7 +1467,7 @@ Model Info:
|
|||
try:
|
||||
await self._flush_digest_buckets()
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.debug(f"Error flushing digest buckets: {e!s}")
|
||||
verbose_proxy_logger.debug("Error flushing digest buckets: %s", e)
|
||||
await self.flush_queue()
|
||||
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -1502,7 +1502,7 @@ Model Info:
|
|||
)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.error(
|
||||
f"[Non-Blocking Error] Slack Alerting: Got error in logging LLM deployment latency: {e!s}"
|
||||
"[Non-Blocking Error] Slack Alerting: Got error in logging LLM deployment latency: %s", e
|
||||
)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -1522,7 +1522,7 @@ Model Info:
|
|||
)
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Exception raises -{e!s}")
|
||||
verbose_logger.debug("Exception raises -%s", e)
|
||||
|
||||
if isinstance(kwargs.get("exception", ""), APIError):
|
||||
if "outage_alerts" in self.alert_types:
|
||||
|
|
@ -1662,9 +1662,9 @@ Model Info:
|
|||
)
|
||||
|
||||
except ValueError as ve:
|
||||
verbose_proxy_logger.error(f"Invalid time range format: {ve}")
|
||||
verbose_proxy_logger.error("Invalid time range format: %s", ve)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.error(f"Error sending spend report: {e}")
|
||||
verbose_proxy_logger.error("Error sending spend report: %s", e)
|
||||
|
||||
async def send_monthly_spend_report(self):
|
||||
""" """
|
||||
|
|
|
|||
|
|
@ -143,8 +143,8 @@ class AnthropicCacheControlHook(CustomPromptManagement):
|
|||
|
||||
if limit_reached:
|
||||
verbose_logger.warning(
|
||||
f"AnthropicCacheControlHook: Reached the Anthropic limit of "
|
||||
f"{MAX_CACHE_CONTROL_BLOCKS} cache_control blocks. Skipping further injection."
|
||||
"AnthropicCacheControlHook: Reached the Anthropic limit of %s cache_control blocks. Skipping further injection.",
|
||||
MAX_CACHE_CONTROL_BLOCKS,
|
||||
)
|
||||
|
||||
return messages
|
||||
|
|
@ -174,8 +174,10 @@ class AnthropicCacheControlHook(CustomPromptManagement):
|
|||
return [targetted_index]
|
||||
|
||||
verbose_logger.warning(
|
||||
f"AnthropicCacheControlHook: Provided index {original_index} is out of bounds for message list of length {len(messages)}. "
|
||||
f"Targeted index was {targetted_index}. Skipping cache control injection for this point."
|
||||
"AnthropicCacheControlHook: Provided index %s is out of bounds for message list of length %s. Targeted index was %s. Skipping cache control injection for this point.",
|
||||
original_index,
|
||||
len(messages),
|
||||
targetted_index,
|
||||
)
|
||||
return []
|
||||
|
||||
|
|
|
|||
|
|
@ -185,9 +185,9 @@ class ArgillaLogger(CustomBatchLogger):
|
|||
)
|
||||
|
||||
if response.status_code >= 300:
|
||||
verbose_logger.error(f"Argilla Error: {response.status_code} - {response.text}")
|
||||
verbose_logger.error("Argilla Error: %s - %s", response.status_code, response.text)
|
||||
else:
|
||||
verbose_logger.debug(f"Batch of {len(self.log_queue)} runs successfully created")
|
||||
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
|
||||
|
||||
self.log_queue.clear()
|
||||
except Exception:
|
||||
|
|
@ -204,7 +204,7 @@ class ArgillaLogger(CustomBatchLogger):
|
|||
random_sample = random.random()
|
||||
if random_sample > sampling_rate:
|
||||
verbose_logger.info(
|
||||
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
|
||||
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
|
||||
)
|
||||
return # Skip logging
|
||||
verbose_logger.debug(
|
||||
|
|
@ -217,7 +217,7 @@ class ArgillaLogger(CustomBatchLogger):
|
|||
return
|
||||
|
||||
self.log_queue.append(data)
|
||||
verbose_logger.debug(f"Langsmith, event added to queue. Will flush in {self.flush_interval} seconds...")
|
||||
verbose_logger.debug("Langsmith, event added to queue. Will flush in %s seconds...", self.flush_interval)
|
||||
|
||||
if len(self.log_queue) >= self.batch_size:
|
||||
self._send_batch()
|
||||
|
|
@ -231,7 +231,7 @@ class ArgillaLogger(CustomBatchLogger):
|
|||
random_sample = random.random()
|
||||
if random_sample > sampling_rate:
|
||||
verbose_logger.info(
|
||||
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
|
||||
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
|
||||
)
|
||||
return # Skip logging
|
||||
verbose_logger.debug(
|
||||
|
|
@ -272,7 +272,7 @@ class ArgillaLogger(CustomBatchLogger):
|
|||
random_sample = random.random()
|
||||
if random_sample > sampling_rate:
|
||||
verbose_logger.info(
|
||||
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
|
||||
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
|
||||
)
|
||||
return # Skip logging
|
||||
verbose_logger.info("Langsmith Failure Event Logging!")
|
||||
|
|
@ -325,7 +325,7 @@ class ArgillaLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
|
||||
if response.status_code >= 300:
|
||||
verbose_logger.error(f"Argilla Error: {response.status_code} - {response.text}")
|
||||
verbose_logger.error("Argilla Error: %s - %s", response.status_code, response.text)
|
||||
else:
|
||||
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
|
||||
except httpx.HTTPStatusError:
|
||||
|
|
|
|||
|
|
@ -461,7 +461,7 @@ def set_attributes(span: "Span", kwargs, response_obj, attributes: type[BaseLLMO
|
|||
_set_response_attributes(span=span, response_obj=response_obj_for_attrs)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"[Arize/Phoenix] Failed to set OpenInference span attributes: {e}")
|
||||
verbose_logger.error("[Arize/Phoenix] Failed to set OpenInference span attributes: %s", e)
|
||||
if hasattr(span, "record_exception"):
|
||||
span.record_exception(e)
|
||||
|
||||
|
|
|
|||
|
|
@ -169,7 +169,7 @@ class ArizeLogger(OpenTelemetry):
|
|||
except Exception as e:
|
||||
return {
|
||||
"status": "unhealthy",
|
||||
"error_message": f"Arize health check failed: {e!s}",
|
||||
"error_message": f"Arize health check failed: {e}",
|
||||
}
|
||||
|
||||
def construct_dynamic_otel_headers(
|
||||
|
|
|
|||
|
|
@ -425,7 +425,7 @@ class ArizePhoenixLogger(OpenTelemetry): # type: ignore
|
|||
endpoint = "http://localhost:6006/v1/traces"
|
||||
protocol = "otlp_http"
|
||||
verbose_logger.debug(
|
||||
f"No PHOENIX_COLLECTOR_ENDPOINT found, using default local Phoenix endpoint: {endpoint}"
|
||||
"No PHOENIX_COLLECTOR_ENDPOINT found, using default local Phoenix endpoint: %s", endpoint
|
||||
)
|
||||
|
||||
otlp_auth_headers = None
|
||||
|
|
|
|||
|
|
@ -339,7 +339,7 @@ class ArizePhoenixPromptManager(CustomPromptManagement):
|
|||
# Log error but don't fail the call
|
||||
import litellm
|
||||
|
||||
litellm._logging.verbose_proxy_logger.error(f"Error in Arize Phoenix prompt pre_call_hook: {e}")
|
||||
litellm._logging.verbose_proxy_logger.error("Error in Arize Phoenix prompt pre_call_hook: %s", e)
|
||||
return messages, litellm_params
|
||||
|
||||
def get_available_prompts(self) -> list[str]:
|
||||
|
|
|
|||
|
|
@ -203,7 +203,7 @@ class AzureSentinelLogger(CustomBatchLogger):
|
|||
await self.async_send_batch()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Azure Sentinel Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Azure Sentinel Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
"""
|
||||
|
|
@ -233,7 +233,7 @@ class AzureSentinelLogger(CustomBatchLogger):
|
|||
await self.async_send_batch()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Azure Sentinel Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Azure Sentinel Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_log_audit_log_event(self, audit_log: StandardAuditLogPayload) -> None:
|
||||
"""
|
||||
|
|
@ -256,7 +256,7 @@ class AzureSentinelLogger(CustomBatchLogger):
|
|||
await self.async_send_audit_batch()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Azure Sentinel Audit Log Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Azure Sentinel Audit Log Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_send_batch(self):
|
||||
"""
|
||||
|
|
@ -323,7 +323,7 @@ class AzureSentinelLogger(CustomBatchLogger):
|
|||
)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Azure Sentinel Error sending batch API - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Azure Sentinel Error sending batch API - %s\n%s", e, traceback.format_exc())
|
||||
finally:
|
||||
log_queue.clear()
|
||||
|
||||
|
|
|
|||
|
|
@ -54,7 +54,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
super().__init__(**kwargs, flush_lock=self.flush_lock)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(
|
||||
f"AzureBlobStorageLogger: Got exception on init AzureBlobStorageLogger client {e!s}"
|
||||
"AzureBlobStorageLogger: Got exception on init AzureBlobStorageLogger client %s", e
|
||||
)
|
||||
raise e
|
||||
|
||||
|
|
@ -79,7 +79,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
self.log_queue.append(standard_logging_payload)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"AzureBlobStorageLogger Layer Error - {e!s}")
|
||||
verbose_logger.exception("AzureBlobStorageLogger Layer Error - %s", e)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
"""
|
||||
|
|
@ -101,7 +101,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
|
||||
self.log_queue.append(standard_logging_payload)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"AzureBlobStorageLogger Layer Error - {e!s}")
|
||||
verbose_logger.exception("AzureBlobStorageLogger Layer Error - %s", e)
|
||||
|
||||
async def async_send_batch(self):
|
||||
"""
|
||||
|
|
@ -124,7 +124,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
await self.async_upload_payload_to_azure_blob_storage(payload=payload)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"AzureBlobStorageLogger Error sending batch API - {e!s}")
|
||||
verbose_logger.exception("AzureBlobStorageLogger Error sending batch API - %s", e)
|
||||
|
||||
async def async_upload_payload_to_azure_blob_storage(self, payload: StandardLoggingPayload):
|
||||
"""
|
||||
|
|
@ -150,16 +150,16 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
await self._append_data(async_client, base_url, json_payload)
|
||||
await self._flush_data(async_client, base_url, len(payload_bytes))
|
||||
|
||||
verbose_logger.debug(f"Successfully uploaded log to Azure Blob Storage: {filename}")
|
||||
verbose_logger.debug("Successfully uploaded log to Azure Blob Storage: %s", filename)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error uploading to Azure Blob Storage: {e!s}")
|
||||
verbose_logger.exception("Error uploading to Azure Blob Storage: %s", e)
|
||||
raise e
|
||||
|
||||
async def _create_file(self, client: AsyncHTTPHandler, base_url: str):
|
||||
"""Helper method to create the file resource"""
|
||||
try:
|
||||
verbose_logger.debug(f"Creating file resource at: {base_url}")
|
||||
verbose_logger.debug("Creating file resource at: %s", base_url)
|
||||
headers = {
|
||||
"x-ms-version": AZURE_STORAGE_MSFT_VERSION,
|
||||
"Content-Length": "0",
|
||||
|
|
@ -169,13 +169,13 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
verbose_logger.debug("Successfully created file resource")
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error creating file resource: {e!s}")
|
||||
verbose_logger.exception("Error creating file resource: %s", e)
|
||||
raise
|
||||
|
||||
async def _append_data(self, client: AsyncHTTPHandler, base_url: str, json_payload: str):
|
||||
"""Helper method to append data to the file"""
|
||||
try:
|
||||
verbose_logger.debug(f"Appending data to file: {base_url}")
|
||||
verbose_logger.debug("Appending data to file: %s", base_url)
|
||||
headers = {
|
||||
"x-ms-version": AZURE_STORAGE_MSFT_VERSION,
|
||||
"Content-Type": "application/json",
|
||||
|
|
@ -189,13 +189,13 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
verbose_logger.debug("Successfully appended data")
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error appending data: {e!s}")
|
||||
verbose_logger.exception("Error appending data: %s", e)
|
||||
raise
|
||||
|
||||
async def _flush_data(self, client: AsyncHTTPHandler, base_url: str, position: int):
|
||||
"""Helper method to flush the data"""
|
||||
try:
|
||||
verbose_logger.debug(f"Flushing data at position {position}")
|
||||
verbose_logger.debug("Flushing data at position %s", position)
|
||||
headers = {
|
||||
"x-ms-version": AZURE_STORAGE_MSFT_VERSION,
|
||||
"Content-Length": "0",
|
||||
|
|
@ -205,7 +205,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
verbose_logger.debug("Successfully flushed data")
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error flushing data: {e!s}")
|
||||
verbose_logger.exception("Error flushing data: %s", e)
|
||||
raise
|
||||
|
||||
####### Helper methods to managing Authentication to Azure Storage #######
|
||||
|
|
@ -229,7 +229,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
)
|
||||
# Token typically expires in 1 hour
|
||||
self.token_expiry = datetime.now() + timedelta(hours=1)
|
||||
verbose_logger.debug(f"New token will expire at {self.token_expiry}")
|
||||
verbose_logger.debug("New token will expire at %s", self.token_expiry)
|
||||
|
||||
def get_azure_ad_token_from_azure_storage(
|
||||
self,
|
||||
|
|
@ -324,7 +324,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
# check if the directory exists
|
||||
if not await directory_client.exists():
|
||||
await directory_client.create_directory()
|
||||
verbose_logger.debug(f"Created directory: {today}")
|
||||
verbose_logger.debug("Created directory: %s", today)
|
||||
|
||||
# Create a file client
|
||||
file_name = f"{payload.get('id') or str(uuid.uuid4())}.json"
|
||||
|
|
@ -342,7 +342,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
|
|||
# Flush the content to finalize the file
|
||||
await file_client.flush_data(position=len(content), offset=0)
|
||||
|
||||
verbose_logger.debug(f"Successfully uploaded and wrote to {today}/{file_name}")
|
||||
verbose_logger.debug("Successfully uploaded and wrote to %s/%s", today, file_name)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error occurred: {e!s}")
|
||||
verbose_logger.exception("Error occurred: %s", e)
|
||||
|
|
|
|||
|
|
@ -320,7 +320,7 @@ class BitBucketPromptManager(CustomPromptManagement):
|
|||
# Log error but don't fail the call
|
||||
import litellm
|
||||
|
||||
litellm._logging.verbose_proxy_logger.error(f"Error in BitBucket prompt pre_call_hook: {e}")
|
||||
litellm._logging.verbose_proxy_logger.error("Error in BitBucket prompt pre_call_hook: %s", e)
|
||||
return messages, litellm_params
|
||||
|
||||
def _parse_prompt_to_messages(self, prompt_content: str) -> list[AllMessageValues]:
|
||||
|
|
|
|||
|
|
@ -89,7 +89,7 @@ def _mock_http_handler_post(
|
|||
"""Monkey-patched HTTPHandler.post that intercepts Braintrust calls with endpoint-specific responses."""
|
||||
# Only mock Braintrust API calls
|
||||
if isinstance(url, str) and _is_braintrust_url(url):
|
||||
verbose_logger.info(f"[BRAINTRUST MOCK] POST to {url}")
|
||||
verbose_logger.info("[BRAINTRUST MOCK] POST to %s", url)
|
||||
time.sleep(_MOCK_LATENCY_SECONDS)
|
||||
# Return appropriate mock response based on endpoint
|
||||
if "/project" in url:
|
||||
|
|
|
|||
|
|
@ -38,7 +38,7 @@ class CloudZeroLogger(CustomLogger):
|
|||
self.connection_id = connection_id or os.getenv("CLOUDZERO_CONNECTION_ID")
|
||||
self.timezone = timezone or os.getenv("CLOUDZERO_TIMEZONE", "UTC")
|
||||
verbose_logger.debug(
|
||||
f"CloudZero Logger initialized with connection ID: {self.connection_id}, timezone: {self.timezone}"
|
||||
"CloudZero Logger initialized with connection ID: %s, timezone: %s", self.connection_id, self.timezone
|
||||
)
|
||||
|
||||
async def initialize_cloudzero_export_job(self):
|
||||
|
|
@ -130,7 +130,7 @@ class CloudZeroLogger(CustomLogger):
|
|||
verbose_logger.debug("CloudZero Logger: No usage data found to export")
|
||||
return
|
||||
|
||||
verbose_logger.debug(f"CloudZero Logger: Processing {len(data)} records")
|
||||
verbose_logger.debug("CloudZero Logger: Processing %s records", len(data))
|
||||
|
||||
# Transform data to CloudZero CBF format
|
||||
transformer = CBFTransformer()
|
||||
|
|
@ -147,13 +147,13 @@ class CloudZeroLogger(CustomLogger):
|
|||
user_timezone=self.timezone,
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"CloudZero Logger: Transmitting {len(cbf_data)} records to CloudZero")
|
||||
verbose_logger.debug("CloudZero Logger: Transmitting %s records to CloudZero", len(cbf_data))
|
||||
streamer.send_batched(cbf_data, operation=operation)
|
||||
|
||||
verbose_logger.debug(f"CloudZero Logger: Successfully exported {len(cbf_data)} records to CloudZero")
|
||||
verbose_logger.debug("CloudZero Logger: Successfully exported %s records to CloudZero", len(cbf_data))
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"CloudZero Logger: Error exporting usage data: {e!s}")
|
||||
verbose_logger.error("CloudZero Logger: Error exporting usage data: %s", e)
|
||||
raise
|
||||
|
||||
async def dry_run_export_usage_data(self, limit: int | None = 10000):
|
||||
|
|
@ -191,7 +191,7 @@ class CloudZeroLogger(CustomLogger):
|
|||
},
|
||||
}
|
||||
|
||||
verbose_logger.debug(f"CloudZero Dry Run: Processing {len(data)} records...")
|
||||
verbose_logger.debug("CloudZero Dry Run: Processing %s records...", len(data))
|
||||
|
||||
# Convert usage data to dict format for response
|
||||
usage_data_sample = data.head(50).to_dicts() # Return first 50 rows
|
||||
|
|
@ -229,7 +229,7 @@ class CloudZeroLogger(CustomLogger):
|
|||
)
|
||||
total_tokens = sum(record.get("usage/amount", 0) for record in cbf_data_dict)
|
||||
|
||||
verbose_logger.debug(f"CloudZero Logger: Dry run completed for {len(cbf_data)} records")
|
||||
verbose_logger.debug("CloudZero Logger: Dry run completed for %s records", len(cbf_data))
|
||||
|
||||
return {
|
||||
"usage_data": usage_data_sample,
|
||||
|
|
@ -244,8 +244,8 @@ class CloudZeroLogger(CustomLogger):
|
|||
}
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"CloudZero Logger: Error in dry run export: {e!s}")
|
||||
verbose_logger.error(f"CloudZero Dry Run Error: {e!s}")
|
||||
verbose_logger.error("CloudZero Logger: Error in dry run export: %s", e)
|
||||
verbose_logger.error("CloudZero Dry Run Error: %s", e)
|
||||
raise
|
||||
|
||||
def _display_cbf_data_on_screen(self, cbf_data):
|
||||
|
|
|
|||
|
|
@ -98,4 +98,4 @@ class LiteLLMDatabase:
|
|||
# This prevents schema mismatch errors when data types vary across rows
|
||||
return pl.DataFrame(db_response, infer_schema_length=None)
|
||||
except Exception as e:
|
||||
raise Exception(f"Error retrieving usage data: {e!s}")
|
||||
raise Exception(f"Error retrieving usage data: {e}")
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ class CustomBatchLogger(CustomLogger):
|
|||
async def periodic_flush(self):
|
||||
while True:
|
||||
await asyncio.sleep(self.flush_interval)
|
||||
verbose_logger.debug(f"CustomLogger periodic flush after {self.flush_interval} seconds")
|
||||
verbose_logger.debug("CustomLogger periodic flush after %s seconds", self.flush_interval)
|
||||
await self.flush_queue()
|
||||
|
||||
async def flush_queue(self):
|
||||
|
|
|
|||
|
|
@ -864,7 +864,7 @@ class CustomGuardrail(CustomLogger):
|
|||
|
||||
if premium_user is not True:
|
||||
verbose_logger.warning(
|
||||
f"Trying to use premium guardrail without premium user {CommonProxyErrors.not_premium_user.value}"
|
||||
"Trying to use premium guardrail without premium user %s", CommonProxyErrors.not_premium_user.value
|
||||
)
|
||||
return False
|
||||
return True
|
||||
|
|
@ -1028,7 +1028,7 @@ class CustomGuardrail(CustomLogger):
|
|||
else:
|
||||
guardrail_response = "allow"
|
||||
|
||||
verbose_logger.debug(f"Guardrail response: {response}")
|
||||
verbose_logger.debug("Guardrail response: %s", response)
|
||||
|
||||
self.add_standard_logging_guardrail_information_to_request_data(
|
||||
guardrail_json_response=guardrail_response,
|
||||
|
|
|
|||
|
|
@ -915,19 +915,19 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
|
|||
|
||||
for callback_obj in all_callbacks:
|
||||
if hasattr(callback_obj, "increment_callback_logging_failure"):
|
||||
verbose_logger.debug(f"Incrementing callback failure metric for {callback_name}")
|
||||
verbose_logger.debug("Incrementing callback failure metric for %s", callback_name)
|
||||
callback_obj.increment_callback_logging_failure(callback_name=callback_name) # type: ignore
|
||||
return
|
||||
|
||||
verbose_logger.debug(
|
||||
f"No callback with increment_callback_logging_failure method found for {callback_name}. "
|
||||
"Ensure 'prometheus' is in your callbacks config."
|
||||
"No callback with increment_callback_logging_failure method found for %s. Ensure 'prometheus' is in your callbacks config.",
|
||||
callback_name,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
from litellm._logging import verbose_logger
|
||||
|
||||
verbose_logger.debug(f"Error in handle_callback_failure for {callback_name}: {e!s}")
|
||||
verbose_logger.debug("Error in handle_callback_failure for %s: %s", callback_name, e)
|
||||
|
||||
async def _strip_base64_from_messages(
|
||||
self,
|
||||
|
|
@ -946,7 +946,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
|
|||
"""
|
||||
raw_messages: Any = payload.get("messages", [])
|
||||
messages: list[Any] = raw_messages if isinstance(raw_messages, list) else []
|
||||
verbose_logger.debug(f"[CustomLogger] Stripping base64 from {len(messages)} messages")
|
||||
verbose_logger.debug("[CustomLogger] Stripping base64 from %s messages", len(messages))
|
||||
|
||||
if messages:
|
||||
payload["messages"] = self._process_messages(messages=messages, max_depth=max_depth)
|
||||
|
|
@ -958,7 +958,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
|
|||
if isinstance(content, list):
|
||||
total_items += len(content)
|
||||
|
||||
verbose_logger.debug(f"[CustomLogger] Completed base64 strip; retained {total_items} content items")
|
||||
verbose_logger.debug("[CustomLogger] Completed base64 strip; retained %s content items", total_items)
|
||||
return payload
|
||||
|
||||
def _strip_base64_from_messages_sync(
|
||||
|
|
@ -978,7 +978,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
|
|||
"""
|
||||
raw_messages: Any = payload.get("messages", [])
|
||||
messages: list[Any] = raw_messages if isinstance(raw_messages, list) else []
|
||||
verbose_logger.debug(f"[CustomLogger] Stripping base64 from {len(messages)} messages")
|
||||
verbose_logger.debug("[CustomLogger] Stripping base64 from %s messages", len(messages))
|
||||
|
||||
if messages:
|
||||
payload["messages"] = self._process_messages(messages=messages, max_depth=max_depth)
|
||||
|
|
@ -990,7 +990,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
|
|||
if isinstance(content, list):
|
||||
total_items += len(content)
|
||||
|
||||
verbose_logger.debug(f"[CustomLogger] Completed base64 strip; retained {total_items} content items")
|
||||
verbose_logger.debug("[CustomLogger] Completed base64 strip; retained %s content items", total_items)
|
||||
return payload
|
||||
|
||||
def _redact_base64(
|
||||
|
|
@ -1001,12 +1001,12 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
|
|||
) -> Any:
|
||||
"""Recursively redact inline base64 from any nested structure with a max recursion depth limit."""
|
||||
if depth > max_depth:
|
||||
verbose_logger.warning(f"[CustomLogger] Max recursion depth {max_depth} reached while redacting base64")
|
||||
verbose_logger.warning("[CustomLogger] Max recursion depth %s reached while redacting base64", max_depth)
|
||||
return "[MAX_DEPTH_REACHED]"
|
||||
|
||||
if isinstance(value, str):
|
||||
if _BASE64_INLINE_PATTERN.search(value):
|
||||
verbose_logger.debug(f"[CustomLogger] Redacted inline base64 string: {value[:40]}...")
|
||||
verbose_logger.debug("[CustomLogger] Redacted inline base64 string: %s...", value[:40])
|
||||
return _BASE64_INLINE_PATTERN.sub("[BASE64_REDACTED]", value)
|
||||
return value
|
||||
|
||||
|
|
|
|||
|
|
@ -237,7 +237,7 @@ class CustomSecretManager(BaseSecretManager):
|
|||
Returns:
|
||||
True if the secret manager is healthy, False otherwise
|
||||
"""
|
||||
verbose_logger.debug(f"Health check not implemented for {self.secret_manager_name}")
|
||||
verbose_logger.debug("Health check not implemented for %s", self.secret_manager_name)
|
||||
return True
|
||||
|
||||
def __repr__(self) -> str:
|
||||
|
|
|
|||
|
|
@ -171,7 +171,7 @@ class DataDogLogger(
|
|||
batch_size=_resolve_dd_batch_size(),
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog: Got exception on init Datadog client {e!s}")
|
||||
verbose_logger.exception("Datadog: Got exception on init Datadog client %s", e)
|
||||
raise e
|
||||
|
||||
def _get_datadog_params(self) -> dict:
|
||||
|
|
@ -210,7 +210,7 @@ class DataDogLogger(
|
|||
self.DD_API_KEY = dd_api_key or (
|
||||
os.getenv("DD_API_KEY") if allow_env_credentials else None
|
||||
) # Optional when using agent
|
||||
verbose_logger.debug(f"Datadog: Using DD Agent at {self.intake_url}")
|
||||
verbose_logger.debug("Datadog: Using DD Agent at %s", self.intake_url)
|
||||
|
||||
def _configure_dd_direct_api(
|
||||
self,
|
||||
|
|
@ -257,7 +257,7 @@ class DataDogLogger(
|
|||
await self._log_async_event(kwargs, response_obj, start_time, end_time)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Datadog Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
|
|
@ -265,7 +265,7 @@ class DataDogLogger(
|
|||
await self._log_async_event(kwargs, response_obj, start_time, end_time)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Datadog Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_post_call_failure_hook(
|
||||
self,
|
||||
|
|
@ -340,7 +340,7 @@ class DataDogLogger(
|
|||
if len(self.log_queue) >= self.batch_size:
|
||||
await self.flush_queue()
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog: async_post_call_failure_hook - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Datadog: async_post_call_failure_hook - %s\n%s", e, traceback.format_exc())
|
||||
return None
|
||||
|
||||
async def async_send_batch(self):
|
||||
|
|
@ -376,11 +376,11 @@ class DataDogLogger(
|
|||
self.log_queue = undelivered + self.log_queue
|
||||
|
||||
if self.is_mock_mode:
|
||||
verbose_logger.debug(f"[DATADOG MOCK] Batch of {len(batch_to_send)} events successfully mocked")
|
||||
verbose_logger.debug("[DATADOG MOCK] Batch of %s events successfully mocked", len(batch_to_send))
|
||||
|
||||
except Exception as e:
|
||||
self.log_queue = batch_to_send + self.log_queue
|
||||
verbose_logger.exception(f"Datadog Error sending batch API - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Datadog Error sending batch API - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def _send_with_413_split(self, batch: list) -> list:
|
||||
"""
|
||||
|
|
@ -411,7 +411,7 @@ class DataDogLogger(
|
|||
if isinstance(e, MaskedHTTPStatusError) and e.status_code == 413:
|
||||
response = e.response
|
||||
else:
|
||||
verbose_logger.exception(f"Datadog Error sending batch API - {e!s}")
|
||||
verbose_logger.exception("Datadog Error sending batch API - %s", e)
|
||||
return self._undelivered(chunk, pending)
|
||||
|
||||
if response.status_code == 413:
|
||||
|
|
@ -515,7 +515,7 @@ class DataDogLogger(
|
|||
)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Datadog Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def _log_async_event(self, kwargs, response_obj, start_time, end_time):
|
||||
dd_payload = self.create_datadog_logging_payload(
|
||||
|
|
@ -526,7 +526,7 @@ class DataDogLogger(
|
|||
)
|
||||
|
||||
self.log_queue.append(dd_payload)
|
||||
verbose_logger.debug(f"Datadog, event added to queue. Will flush in {self.flush_interval} seconds...")
|
||||
verbose_logger.debug("Datadog, event added to queue. Will flush in %s seconds...", self.flush_interval)
|
||||
|
||||
if len(self.log_queue) >= self.batch_size:
|
||||
await self.flush_queue()
|
||||
|
|
@ -653,7 +653,7 @@ class DataDogLogger(
|
|||
self.log_queue.append(_dd_payload)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog: Logger - Exception in async_service_failure_hook: {e}")
|
||||
verbose_logger.exception("Datadog: Logger - Exception in async_service_failure_hook: %s", e)
|
||||
|
||||
async def async_service_success_hook(
|
||||
self,
|
||||
|
|
@ -692,7 +692,7 @@ class DataDogLogger(
|
|||
self.log_queue.append(_dd_payload)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog: Logger - Exception in async_service_failure_hook: {e}")
|
||||
verbose_logger.exception("Datadog: Logger - Exception in async_service_failure_hook: %s", e)
|
||||
|
||||
def _create_v0_logging_payload(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -84,7 +84,7 @@ class DatadogCostManagementLogger(CustomBatchLogger):
|
|||
await self.async_send_batch()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog Cost Management: Error in async_log_success_event: {e!s}")
|
||||
verbose_logger.exception("Datadog Cost Management: Error in async_log_success_event: %s", e)
|
||||
|
||||
async def async_send_batch(self):
|
||||
if not self.log_queue:
|
||||
|
|
@ -104,7 +104,7 @@ class DatadogCostManagementLogger(CustomBatchLogger):
|
|||
await self._upload_to_datadog(aggregated_entries)
|
||||
except Exception as e:
|
||||
self.log_queue = batch_to_send + self.log_queue
|
||||
verbose_logger.exception(f"Datadog Cost Management: Error in async_send_batch: {e!s}")
|
||||
verbose_logger.exception("Datadog Cost Management: Error in async_send_batch: %s", e)
|
||||
|
||||
def _aggregate_costs(self, logs: list[StandardLoggingPayload]) -> list[DatadogFOCUSCostEntry]:
|
||||
"""
|
||||
|
|
@ -159,7 +159,7 @@ class DatadogCostManagementLogger(CustomBatchLogger):
|
|||
aggregator[key]["BilledCost"] += cost
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error processing log for cost aggregation: {e}")
|
||||
verbose_logger.warning("Error processing log for cost aggregation: %s", e)
|
||||
continue
|
||||
|
||||
return list(aggregator.values())
|
||||
|
|
@ -254,5 +254,5 @@ class DatadogCostManagementLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
|
||||
verbose_logger.debug(
|
||||
f"Datadog Cost Management: Uploaded {len(payload)} cost entries. Status: {response.status_code}"
|
||||
"Datadog Cost Management: Uploaded %s cost entries. Status: %s", len(payload), response.status_code
|
||||
)
|
||||
|
|
|
|||
|
|
@ -89,7 +89,7 @@ class DataDogLLMObsLogger(CustomBatchLogger):
|
|||
kwargs.update(dict_datadog_llm_obs_params)
|
||||
CustomBatchLogger.__init__(self, **kwargs, flush_lock=self.flush_lock)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"DataDogLLMObs: Error initializing - {e!s}")
|
||||
verbose_logger.exception("DataDogLLMObs: Error initializing - %s", e)
|
||||
raise e
|
||||
|
||||
def _configure_dd_agent(self, dd_agent_host: str):
|
||||
|
|
@ -103,7 +103,7 @@ class DataDogLLMObsLogger(CustomBatchLogger):
|
|||
agent_port = os.getenv("LITELLM_DD_LLM_OBS_PORT", "8126")
|
||||
self.DD_SITE = "localhost" # Not used for URL construction in agent mode
|
||||
self.intake_url = f"http://{dd_agent_host}:{agent_port}/api/intake/llm-obs/v1/trace/spans"
|
||||
verbose_logger.debug(f"DataDogLLMObs: Using DD Agent at {self.intake_url}")
|
||||
verbose_logger.debug("DataDogLLMObs: Using DD Agent at %s", self.intake_url)
|
||||
|
||||
def _configure_dd_direct_api(self):
|
||||
"""
|
||||
|
|
@ -137,34 +137,34 @@ class DataDogLLMObsLogger(CustomBatchLogger):
|
|||
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
verbose_logger.debug(f"DataDogLLMObs: Logging success event for model {kwargs.get('model', 'unknown')}")
|
||||
verbose_logger.debug("DataDogLLMObs: Logging success event for model %s", kwargs.get("model", "unknown"))
|
||||
payload = self.create_llm_obs_payload(kwargs, start_time, end_time)
|
||||
verbose_logger.debug(f"DataDogLLMObs: Payload: {payload}")
|
||||
verbose_logger.debug("DataDogLLMObs: Payload: %s", payload)
|
||||
self.log_queue.append(payload)
|
||||
|
||||
if len(self.log_queue) >= self.batch_size:
|
||||
await self.async_send_batch()
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"DataDogLLMObs: Error logging success event - {e!s}")
|
||||
verbose_logger.exception("DataDogLLMObs: Error logging success event - %s", e)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
verbose_logger.debug(f"DataDogLLMObs: Logging failure event for model {kwargs.get('model', 'unknown')}")
|
||||
verbose_logger.debug("DataDogLLMObs: Logging failure event for model %s", kwargs.get("model", "unknown"))
|
||||
payload = self.create_llm_obs_payload(kwargs, start_time, end_time)
|
||||
verbose_logger.debug(f"DataDogLLMObs: Payload: {payload}")
|
||||
verbose_logger.debug("DataDogLLMObs: Payload: %s", payload)
|
||||
self.log_queue.append(payload)
|
||||
|
||||
if len(self.log_queue) >= self.batch_size:
|
||||
await self.async_send_batch()
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"DataDogLLMObs: Error logging failure event - {e!s}")
|
||||
verbose_logger.exception("DataDogLLMObs: Error logging failure event - %s", e)
|
||||
|
||||
async def async_send_batch(self):
|
||||
try:
|
||||
if not self.log_queue:
|
||||
return
|
||||
|
||||
verbose_logger.debug(f"DataDogLLMObs: Flushing {len(self.log_queue)} events")
|
||||
verbose_logger.debug("DataDogLLMObs: Flushing %s events", len(self.log_queue))
|
||||
|
||||
if self.is_mock_mode:
|
||||
verbose_logger.debug("[DATADOG MOCK] Mock mode enabled - API calls will be intercepted")
|
||||
|
|
@ -207,14 +207,14 @@ class DataDogLLMObsLogger(CustomBatchLogger):
|
|||
)
|
||||
|
||||
if self.is_mock_mode:
|
||||
verbose_logger.debug(f"[DATADOG MOCK] Batch of {len(self.log_queue)} events successfully mocked")
|
||||
verbose_logger.debug("[DATADOG MOCK] Batch of %s events successfully mocked", len(self.log_queue))
|
||||
else:
|
||||
verbose_logger.debug(f"DataDogLLMObs: Successfully sent batch - status_code: {response.status_code}")
|
||||
verbose_logger.debug("DataDogLLMObs: Successfully sent batch - status_code: %s", response.status_code)
|
||||
self.log_queue.clear()
|
||||
except httpx.HTTPStatusError as e:
|
||||
verbose_logger.exception(f"DataDogLLMObs: Error sending batch - {e.response.text}")
|
||||
verbose_logger.exception("DataDogLLMObs: Error sending batch - %s", e.response.text)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"DataDogLLMObs: Error sending batch - {e!s}")
|
||||
verbose_logger.exception("DataDogLLMObs: Error sending batch - %s", e)
|
||||
|
||||
def create_llm_obs_payload(self, kwargs: dict, start_time: datetime, end_time: datetime) -> LLMObsPayload:
|
||||
standard_logging_payload: StandardLoggingPayload | None = kwargs.get("standard_logging_object")
|
||||
|
|
@ -613,7 +613,7 @@ class DataDogLLMObsLogger(CustomBatchLogger):
|
|||
try:
|
||||
spend_metrics["user_api_key_spend"] = float(user_api_key_spend)
|
||||
except (ValueError, TypeError):
|
||||
verbose_logger.debug(f"Invalid user_api_key_spend value: {user_api_key_spend}")
|
||||
verbose_logger.debug("Invalid user_api_key_spend value: %s", user_api_key_spend)
|
||||
|
||||
# API key budget reset datetime
|
||||
user_api_key_budget_reset_at = metadata.get("user_api_key_budget_reset_at")
|
||||
|
|
@ -640,10 +640,10 @@ class DataDogLLMObsLogger(CustomBatchLogger):
|
|||
spend_metrics["user_api_key_budget_reset_at"] = iso_string
|
||||
|
||||
# Debug logging to verify the conversion
|
||||
verbose_logger.debug(f"Converted budget_reset_at to ISO format: {iso_string}")
|
||||
verbose_logger.debug("Converted budget_reset_at to ISO format: %s", iso_string)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Error processing budget reset datetime: {e}")
|
||||
verbose_logger.debug(f"Original value: {user_api_key_budget_reset_at}")
|
||||
verbose_logger.debug("Error processing budget reset datetime: %s", e)
|
||||
verbose_logger.debug("Original value: %s", user_api_key_budget_reset_at)
|
||||
|
||||
return spend_metrics
|
||||
|
||||
|
|
@ -707,7 +707,7 @@ class DataDogLLMObsLogger(CustomBatchLogger):
|
|||
|
||||
kv_pairs[f"tool_calls.{idx}.function.arguments"] = json.dumps(function_arguments)
|
||||
except (KeyError, TypeError, ValueError) as e:
|
||||
verbose_logger.debug(f"DataDogLLMObs: Error processing tool call {idx}: {e!s}")
|
||||
verbose_logger.debug("DataDogLLMObs: Error processing tool call %s: %s", idx, e)
|
||||
continue
|
||||
|
||||
return kv_pairs
|
||||
|
|
@ -747,6 +747,6 @@ class DataDogLLMObsLogger(CustomBatchLogger):
|
|||
tool_call_metadata[f"output_{key}"] = value
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"DataDogLLMObs: Error extracting tool call metadata: {e!s}")
|
||||
verbose_logger.debug("DataDogLLMObs: Error extracting tool call metadata: %s", e)
|
||||
|
||||
return tool_call_metadata
|
||||
|
|
|
|||
|
|
@ -180,7 +180,7 @@ class DatadogMetricsLogger(CustomBatchLogger):
|
|||
await self.flush_queue()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog Metrics: Error in async_log_success_event: {e!s}")
|
||||
verbose_logger.exception("Datadog Metrics: Error in async_log_success_event: %s", e)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
|
|
@ -202,7 +202,7 @@ class DatadogMetricsLogger(CustomBatchLogger):
|
|||
await self.flush_queue()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog Metrics: Error in async_log_failure_event: {e!s}")
|
||||
verbose_logger.exception("Datadog Metrics: Error in async_log_failure_event: %s", e)
|
||||
|
||||
async def async_send_batch(self):
|
||||
if not self.log_queue:
|
||||
|
|
@ -214,7 +214,7 @@ class DatadogMetricsLogger(CustomBatchLogger):
|
|||
try:
|
||||
await self._upload_to_datadog(payload_data)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog Metrics: Error in async_send_batch: {e!s}")
|
||||
verbose_logger.exception("Datadog Metrics: Error in async_send_batch: %s", e)
|
||||
raise
|
||||
|
||||
async def _upload_to_datadog(self, payload: DatadogMetricsPayload):
|
||||
|
|
@ -242,7 +242,7 @@ class DatadogMetricsLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
|
||||
verbose_logger.debug(
|
||||
f"Datadog Metrics: Uploaded {len(payload['series'])} metric points. Status: {response.status_code}"
|
||||
"Datadog Metrics: Uploaded %s metric points. Status: %s", len(payload["series"]), response.status_code
|
||||
)
|
||||
|
||||
async def async_health_check(self) -> IntegrationHealthCheckStatus:
|
||||
|
|
|
|||
|
|
@ -23,9 +23,9 @@ def log_retry_error(details):
|
|||
exception = details.get("exception")
|
||||
tries = details.get("tries")
|
||||
if exception:
|
||||
logging.error(f"Confident AI Error: {exception}. Retrying: {tries} time(s)...")
|
||||
logging.error("Confident AI Error: %s. Retrying: %s time(s)...", exception, tries)
|
||||
else:
|
||||
logging.error(f"Retrying: {tries} time(s)...")
|
||||
logging.error("Retrying: %s time(s)...", tries)
|
||||
|
||||
|
||||
class HttpMethods(Enum):
|
||||
|
|
|
|||
|
|
@ -70,7 +70,7 @@ class DyanmoDBLogger:
|
|||
# Assuming log_data is a dictionary with log information
|
||||
response = table.put_item(Item=payload)
|
||||
|
||||
print_verbose(f"Response from DynamoDB:{response!s}")
|
||||
print_verbose(f"Response from DynamoDB:{response}")
|
||||
|
||||
print_verbose(f"DynamoDB Layer Logging - final response object: {response_obj}")
|
||||
return response
|
||||
|
|
|
|||
|
|
@ -128,7 +128,7 @@ class GalileoObserve(CustomLogger):
|
|||
except Exception as e:
|
||||
return IntegrationHealthCheckStatus(
|
||||
status="unhealthy",
|
||||
error_message=f"Galileo health check failed: {e!s}",
|
||||
error_message=f"Galileo health check failed: {e}",
|
||||
)
|
||||
|
||||
async def async_set_galileo_headers(self) -> None:
|
||||
|
|
|
|||
|
|
@ -76,7 +76,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
|
|||
await self.log_queue.put(GCSLogQueueItem(payload=logging_payload, kwargs=kwargs, response_obj=response_obj))
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"GCS Bucket logging error: {e!s}")
|
||||
verbose_logger.exception("GCS Bucket logging error: %s", e)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
|
|
@ -95,7 +95,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
|
|||
await self.log_queue.put(GCSLogQueueItem(payload=logging_payload, kwargs=kwargs, response_obj=response_obj))
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"GCS Bucket logging error: {e!s}")
|
||||
verbose_logger.exception("GCS Bucket logging error: %s", e)
|
||||
|
||||
def _drain_queue_batch(self) -> list[GCSLogQueueItem]:
|
||||
"""
|
||||
|
|
@ -218,7 +218,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
|
|||
except Exception as e:
|
||||
success_count = 0
|
||||
error_count = len(items)
|
||||
verbose_logger.exception(f"GCS Bucket error logging batch payload to GCS bucket: {e!s}")
|
||||
verbose_logger.exception("GCS Bucket error logging batch payload to GCS bucket: %s", e)
|
||||
return (success_count, error_count)
|
||||
|
||||
async def _send_individual_logs(self, items: list[GCSLogQueueItem]) -> None:
|
||||
|
|
@ -255,7 +255,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
|
|||
logging_payload=item["payload"],
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"GCS Bucket error logging individual payload to GCS bucket: {e!s}")
|
||||
verbose_logger.exception("GCS Bucket error logging individual payload to GCS bucket: %s", e)
|
||||
|
||||
async def async_send_batch(self):
|
||||
"""
|
||||
|
|
@ -336,7 +336,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
|
|||
loaded_response = json.loads(response)
|
||||
return loaded_response
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Failed to fetch payload for date {date_str}: {e!s}")
|
||||
verbose_logger.debug("Failed to fetch payload for date %s: %s", date_str, e)
|
||||
continue
|
||||
|
||||
return None
|
||||
|
|
@ -370,7 +370,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
|
|||
"""
|
||||
while True:
|
||||
await asyncio.sleep(self.flush_interval)
|
||||
verbose_logger.debug(f"GCS Bucket periodic flush after {self.flush_interval} seconds")
|
||||
verbose_logger.debug("GCS Bucket periodic flush after %s seconds", self.flush_interval)
|
||||
await self.flush_queue()
|
||||
|
||||
async def async_health_check(self) -> IntegrationHealthCheckStatus:
|
||||
|
|
|
|||
|
|
@ -45,7 +45,7 @@ async def _mock_async_handler_get(self, url, params=None, headers=None, follow_r
|
|||
"""Monkey-patched AsyncHTTPHandler.get that intercepts GCS calls."""
|
||||
# Only mock GCS API calls
|
||||
if isinstance(url, str) and "storage.googleapis.com" in url:
|
||||
verbose_logger.info(f"[GCS MOCK] GET to {url}")
|
||||
verbose_logger.info("[GCS MOCK] GET to %s", url)
|
||||
await asyncio.sleep(_MOCK_LATENCY_SECONDS)
|
||||
# Return a minimal but valid StandardLoggingPayload JSON string as bytes
|
||||
# This matches what GCS returns when downloading with ?alt=media
|
||||
|
|
@ -117,7 +117,7 @@ async def _mock_async_handler_delete(
|
|||
"""Monkey-patched AsyncHTTPHandler.delete that intercepts GCS calls."""
|
||||
# Only mock GCS API calls
|
||||
if isinstance(url, str) and "storage.googleapis.com" in url:
|
||||
verbose_logger.info(f"[GCS MOCK] DELETE to {url}")
|
||||
verbose_logger.info("[GCS MOCK] DELETE to %s", url)
|
||||
await asyncio.sleep(_MOCK_LATENCY_SECONDS)
|
||||
# DELETE returns 204 No Content with empty body (not JSON)
|
||||
return MockResponse(
|
||||
|
|
|
|||
|
|
@ -132,7 +132,7 @@ class GcsPubSubLogger(CustomBatchLogger):
|
|||
await self.async_send_batch()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"PubSub Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("PubSub Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_send_batch(self):
|
||||
"""
|
||||
|
|
@ -142,13 +142,13 @@ class GcsPubSubLogger(CustomBatchLogger):
|
|||
if not self.log_queue:
|
||||
return
|
||||
|
||||
verbose_logger.debug(f"PubSub - about to flush {len(self.log_queue)} events")
|
||||
verbose_logger.debug("PubSub - about to flush %s events", len(self.log_queue))
|
||||
|
||||
for message in self.log_queue:
|
||||
await self.publish_message(message)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"PubSub Error sending batch - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("PubSub Error sending batch - %s\n%s", e, traceback.format_exc())
|
||||
finally:
|
||||
self.log_queue.clear()
|
||||
|
||||
|
|
|
|||
|
|
@ -42,7 +42,7 @@ def load_compatible_callbacks() -> dict:
|
|||
with open(json_path, "r") as f:
|
||||
return json.load(f)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error loading generic_api_compatible_callbacks.json: {e!s}")
|
||||
verbose_logger.warning("Error loading generic_api_compatible_callbacks.json: %s", e)
|
||||
return {}
|
||||
|
||||
|
||||
|
|
@ -124,7 +124,7 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
#########################################################
|
||||
if callback_name:
|
||||
if is_callback_compatible(callback_name):
|
||||
verbose_logger.debug(f"Loading configuration for callback: {callback_name}")
|
||||
verbose_logger.debug("Loading configuration for callback: %s", callback_name)
|
||||
callback_config = get_callback_config(callback_name)
|
||||
|
||||
# Use config from JSON if not explicitly provided
|
||||
|
|
@ -145,7 +145,7 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
log_format = callback_config["log_format"]
|
||||
else:
|
||||
verbose_logger.warning(
|
||||
f"callback_name '{callback_name}' not found in generic_api_compatible_callbacks.json"
|
||||
"callback_name '%s' not found in generic_api_compatible_callbacks.json", callback_name
|
||||
)
|
||||
|
||||
#########################################################
|
||||
|
|
@ -177,7 +177,12 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
self.log_format: LOG_FORMAT_TYPES = log_format or "json_array"
|
||||
|
||||
verbose_logger.debug(
|
||||
f"in init GenericAPILogger, callback_name: {self.callback_name}, endpoint {self.endpoint}, headers {self.headers}, event_types: {self.event_types}, log_format: {self.log_format}"
|
||||
"in init GenericAPILogger, callback_name: %s, endpoint %s, headers %s, event_types: %s, log_format: %s",
|
||||
self.callback_name,
|
||||
self.endpoint,
|
||||
self.headers,
|
||||
self.event_types,
|
||||
self.log_format,
|
||||
)
|
||||
|
||||
#########################################################
|
||||
|
|
@ -214,7 +219,7 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
key, value = item.split("=", 1)
|
||||
headers_dict[key.strip()] = value.strip()
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error parsing headers from environment variables: {e!s}")
|
||||
verbose_logger.warning("Error parsing headers from environment variables: %s", e)
|
||||
|
||||
# 2. Update with litellm generic headers if available
|
||||
if litellm.generic_logger_headers:
|
||||
|
|
@ -308,7 +313,7 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
await self.async_send_batch()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Generic API Logger Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Generic API Logger Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
"""
|
||||
|
|
@ -339,7 +344,7 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
await self.async_send_batch()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Generic API Logger Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Generic API Logger Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_send_batch(self):
|
||||
"""
|
||||
|
|
@ -355,7 +360,7 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
return
|
||||
|
||||
verbose_logger.debug(
|
||||
f"Generic API Logger - about to flush {len(self.log_queue)} events in '{self.log_format}' format"
|
||||
"Generic API Logger - about to flush %s events in '%s' format", len(self.log_queue), self.log_format
|
||||
)
|
||||
|
||||
if self.log_format == "single":
|
||||
|
|
@ -371,11 +376,13 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
# Log results
|
||||
for idx, result in enumerate(responses):
|
||||
if isinstance(result, Exception):
|
||||
verbose_logger.exception(f"Generic API Logger - Error sending log {idx}: {result}")
|
||||
verbose_logger.exception("Generic API Logger - Error sending log %s: %s", idx, result)
|
||||
else:
|
||||
# result is a Response object
|
||||
verbose_logger.debug(
|
||||
f"Generic API Logger - sent log {idx}, status: {result.status_code}" # type: ignore
|
||||
"Generic API Logger - sent log %s, status: %s",
|
||||
idx,
|
||||
result.status_code, # type: ignore
|
||||
)
|
||||
else:
|
||||
# Format the payload based on log_format
|
||||
|
|
@ -390,12 +397,14 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
response = await self._post_with_retries(data=data)
|
||||
|
||||
verbose_logger.debug(
|
||||
f"Generic API Logger - sent batch to {self.endpoint}, "
|
||||
f"status: {response.status_code}, format: {self.log_format}"
|
||||
"Generic API Logger - sent batch to %s, status: %s, format: %s",
|
||||
self.endpoint,
|
||||
response.status_code,
|
||||
self.log_format,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Generic API Logger Error sending batch - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Generic API Logger Error sending batch - %s\n%s", e, traceback.format_exc())
|
||||
finally:
|
||||
self.log_queue.clear()
|
||||
|
||||
|
|
@ -405,7 +414,7 @@ class GenericAPILogger(CustomBatchLogger):
|
|||
|
||||
Returns a dict of the payload to send to the Generic API Endpoint
|
||||
"""
|
||||
verbose_logger.debug(f"GenericAPILogger Logging - Enters logging function for model {kwargs}")
|
||||
verbose_logger.debug("GenericAPILogger Logging - Enters logging function for model %s", kwargs)
|
||||
|
||||
# construct payload to send custom logger
|
||||
# follows the same params as langfuse.py
|
||||
|
|
|
|||
|
|
@ -379,7 +379,7 @@ class GitLabPromptManager(CustomPromptManagement):
|
|||
except Exception as e:
|
||||
import litellm
|
||||
|
||||
litellm._logging.verbose_proxy_logger.error(f"Error in GitLab prompt pre_call_hook: {e}")
|
||||
litellm._logging.verbose_proxy_logger.error("Error in GitLab prompt pre_call_hook: %s", e)
|
||||
return messages, litellm_params
|
||||
|
||||
def _parse_prompt_to_messages(self, prompt_content: str) -> list[AllMessageValues]:
|
||||
|
|
|
|||
|
|
@ -117,7 +117,7 @@ class LagoLogger(CustomLogger):
|
|||
}
|
||||
}
|
||||
|
||||
verbose_logger.debug(f"\033[91mLogged Lago Object:\n{returned_val}\033[0m\n")
|
||||
verbose_logger.debug("\x1b[91mLogged Lago Object:\n%s\x1b[0m\n", returned_val)
|
||||
return returned_val
|
||||
|
||||
def log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -149,7 +149,7 @@ class LagoLogger(CustomLogger):
|
|||
except Exception as e:
|
||||
error_response = getattr(e, "response", None)
|
||||
if error_response is not None and hasattr(error_response, "text"):
|
||||
verbose_logger.debug(f"\nError Message: {error_response.text}")
|
||||
verbose_logger.debug("\nError Message: %s", error_response.text)
|
||||
raise e
|
||||
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -184,8 +184,8 @@ class LagoLogger(CustomLogger):
|
|||
|
||||
response.raise_for_status()
|
||||
|
||||
verbose_logger.debug(f"Logged Lago Object: {response.text}")
|
||||
verbose_logger.debug("Logged Lago Object: %s", response.text)
|
||||
except Exception as e:
|
||||
if response is not None and hasattr(response, "text"):
|
||||
verbose_logger.debug(f"\nError Message: {response.text}")
|
||||
verbose_logger.debug("\nError Message: %s", response.text)
|
||||
raise e
|
||||
|
|
|
|||
|
|
@ -199,7 +199,7 @@ class LangFuseLogger:
|
|||
)
|
||||
langfuse_client = Langfuse(**parameters)
|
||||
litellm.initialized_langfuse_clients += 1
|
||||
verbose_logger.debug(f"Created langfuse client number {litellm.initialized_langfuse_clients}")
|
||||
verbose_logger.debug("Created langfuse client number %s", litellm.initialized_langfuse_clients)
|
||||
return langfuse_client
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -226,9 +226,9 @@ class LangFuseLogger:
|
|||
if metadata_param_key.startswith("langfuse_"):
|
||||
trace_param_key = metadata_param_key.replace("langfuse_", "", 1)
|
||||
if trace_param_key in metadata:
|
||||
verbose_logger.warning(f"Overwriting Langfuse `{trace_param_key}` from request header")
|
||||
verbose_logger.warning("Overwriting Langfuse `%s` from request header", trace_param_key)
|
||||
else:
|
||||
verbose_logger.debug(f"Found Langfuse `{trace_param_key}` in request header")
|
||||
verbose_logger.debug("Found Langfuse `%s` in request header", trace_param_key)
|
||||
metadata[trace_param_key] = proxy_headers.get(metadata_param_key)
|
||||
|
||||
return metadata
|
||||
|
|
@ -256,7 +256,7 @@ class LangFuseLogger:
|
|||
Logs a success or error event on Langfuse
|
||||
"""
|
||||
try:
|
||||
verbose_logger.debug(f"Langfuse Logging - Enters logging function for model {kwargs}")
|
||||
verbose_logger.debug("Langfuse Logging - Enters logging function for model %s", kwargs)
|
||||
|
||||
# set default values for input/output for langfuse logging
|
||||
input = None
|
||||
|
|
@ -295,7 +295,7 @@ class LangFuseLogger:
|
|||
level=level,
|
||||
status_message=status_message,
|
||||
)
|
||||
verbose_logger.debug(f"OUTPUT IN LANGFUSE: {output}; original: {response_obj}")
|
||||
verbose_logger.debug("OUTPUT IN LANGFUSE: %s; original: %s", output, response_obj)
|
||||
trace_id = None
|
||||
generation_id = None
|
||||
if self._is_langfuse_v2():
|
||||
|
|
@ -325,12 +325,12 @@ class LangFuseLogger:
|
|||
input=input,
|
||||
response_obj=response_obj,
|
||||
)
|
||||
verbose_logger.debug(f"Langfuse Layer Logging - final response object: {response_obj}")
|
||||
verbose_logger.debug("Langfuse Layer Logging - final response object: %s", response_obj)
|
||||
verbose_logger.info("Langfuse Layer Logging - logging success")
|
||||
|
||||
return {"trace_id": trace_id, "generation_id": generation_id}
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Langfuse Layer Error(): Exception occured - {e!s}")
|
||||
verbose_logger.exception("Langfuse Layer Error(): Exception occured - %s", e)
|
||||
return {"trace_id": None, "generation_id": None}
|
||||
|
||||
def _get_langfuse_input_output_content(
|
||||
|
|
@ -625,7 +625,7 @@ class LangFuseLogger:
|
|||
trace_params["metadata"] = {"metadata_passed_to_litellm": metadata}
|
||||
|
||||
cost = kwargs.get("response_cost", None)
|
||||
verbose_logger.debug(f"trace: {cost}")
|
||||
verbose_logger.debug("trace: %s", cost)
|
||||
|
||||
clean_metadata["litellm_response_cost"] = cost
|
||||
if standard_logging_object is not None:
|
||||
|
|
@ -780,12 +780,13 @@ class LangFuseLogger:
|
|||
if hasattr(generation_client, "trace_id") and generation_client.trace_id:
|
||||
if generation_client.trace_id != trace_id:
|
||||
verbose_logger.warning(
|
||||
f"Langfuse trace_id mismatch: set {trace_id}, but langfuse returned {generation_client.trace_id}. "
|
||||
"Using our intended trace_id for consistency."
|
||||
"Langfuse trace_id mismatch: set %s, but langfuse returned %s. Using our intended trace_id for consistency.",
|
||||
trace_id,
|
||||
generation_client.trace_id,
|
||||
)
|
||||
return trace_id, generation_id
|
||||
except Exception:
|
||||
verbose_logger.error(f"Langfuse Layer Error - {traceback.format_exc()}")
|
||||
verbose_logger.error("Langfuse Layer Error - %s", traceback.format_exc())
|
||||
return None, None
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -902,7 +903,7 @@ class LangFuseLogger:
|
|||
# For other types, try to apply the function directly
|
||||
return masking_function(data)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Failed to apply masking function: {e}. Returning original data.")
|
||||
verbose_logger.warning("Failed to apply masking function: %s. Returning original data.", e)
|
||||
return data
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -966,7 +967,7 @@ class LangFuseLogger:
|
|||
end_time=guardrail_entry.get("end_time", None), # type: ignore
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"Logged guardrail information as span: {span}")
|
||||
verbose_logger.debug("Logged guardrail information as span: %s", span)
|
||||
span.end()
|
||||
|
||||
|
||||
|
|
@ -1035,7 +1036,7 @@ def _add_prompt_to_generation_params(
|
|||
try:
|
||||
generation_params["prompt"] = langfuse_client.get_prompt(prompt_management_metadata["prompt_id"])
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"[Non-blocking] Langfuse Logger: Error getting prompt client for logging: {e}")
|
||||
verbose_logger.debug("[Non-blocking] Langfuse Logger: Error getting prompt client for logging: %s", e)
|
||||
|
||||
else:
|
||||
generation_params["prompt"] = user_prompt
|
||||
|
|
|
|||
|
|
@ -315,10 +315,10 @@ class LangfuseOtelLogger(OpenTelemetry):
|
|||
if langfuse_host:
|
||||
normalized_host = langfuse_host if langfuse_host.startswith("http") else f"https://{langfuse_host}"
|
||||
endpoint = f"{normalized_host.rstrip('/')}/api/public/otel"
|
||||
verbose_logger.debug(f"Using Langfuse OTEL endpoint from host: {endpoint}")
|
||||
verbose_logger.debug("Using Langfuse OTEL endpoint from host: %s", endpoint)
|
||||
else:
|
||||
endpoint = LANGFUSE_CLOUD_US_ENDPOINT
|
||||
verbose_logger.debug(f"Using Langfuse US cloud endpoint: {endpoint}")
|
||||
verbose_logger.debug("Using Langfuse US cloud endpoint: %s", endpoint)
|
||||
|
||||
auth_header = LangfuseOtelLogger._get_langfuse_authorization_header(
|
||||
public_key=public_key, secret_key=secret_key
|
||||
|
|
|
|||
|
|
@ -317,7 +317,7 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
|
|||
except Exception as e:
|
||||
from litellm._logging import verbose_logger
|
||||
|
||||
verbose_logger.exception(f"Langfuse Layer Error - Exception occurred while logging success event: {e!s}")
|
||||
verbose_logger.exception("Langfuse Layer Error - Exception occurred while logging success event: %s", e)
|
||||
self.handle_callback_failure(callback_name="langfuse")
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -347,5 +347,5 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
|
|||
except Exception as e:
|
||||
from litellm._logging import verbose_logger
|
||||
|
||||
verbose_logger.exception(f"Langfuse Layer Error - Exception occurred while logging failure event: {e!s}")
|
||||
verbose_logger.exception("Langfuse Layer Error - Exception occurred while logging failure event: %s", e)
|
||||
self.handle_callback_failure(callback_name="langfuse")
|
||||
|
|
|
|||
|
|
@ -194,7 +194,7 @@ class LangsmithLogger(CustomBatchLogger):
|
|||
|
||||
fields = self._extract_metadata_fields(metadata, credentials)
|
||||
verbose_logger.debug(
|
||||
f"Langsmith Logging - project_name: {fields['project_name']}, run_name {fields['run_name']}"
|
||||
"Langsmith Logging - project_name: %s, run_name %s", fields["project_name"], fields["run_name"]
|
||||
)
|
||||
|
||||
payload: StandardLoggingPayload | None = kwargs.get("standard_logging_object", None)
|
||||
|
|
@ -244,7 +244,7 @@ class LangsmithLogger(CustomBatchLogger):
|
|||
random_sample = random.random()
|
||||
if random_sample > sampling_rate:
|
||||
verbose_logger.info(
|
||||
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
|
||||
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
|
||||
)
|
||||
return # Skip logging
|
||||
verbose_logger.debug(
|
||||
|
|
@ -267,7 +267,7 @@ class LangsmithLogger(CustomBatchLogger):
|
|||
credentials=credentials,
|
||||
)
|
||||
)
|
||||
verbose_logger.debug(f"Langsmith, event added to queue. Will flush in {self.flush_interval} seconds...")
|
||||
verbose_logger.debug("Langsmith, event added to queue. Will flush in %s seconds...", self.flush_interval)
|
||||
|
||||
if len(self.log_queue) >= self.batch_size:
|
||||
self._send_batch()
|
||||
|
|
@ -282,7 +282,7 @@ class LangsmithLogger(CustomBatchLogger):
|
|||
random_sample = random.random()
|
||||
if random_sample > sampling_rate:
|
||||
verbose_logger.info(
|
||||
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
|
||||
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
|
||||
)
|
||||
return # Skip logging
|
||||
verbose_logger.debug(
|
||||
|
|
@ -321,7 +321,7 @@ class LangsmithLogger(CustomBatchLogger):
|
|||
random_sample = random.random()
|
||||
if random_sample > sampling_rate:
|
||||
verbose_logger.info(
|
||||
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
|
||||
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
|
||||
)
|
||||
return # Skip logging
|
||||
verbose_logger.info("Langsmith Failure Event Logging!")
|
||||
|
|
@ -422,16 +422,16 @@ class LangsmithLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
|
||||
if response.status_code >= 300:
|
||||
verbose_logger.error(f"Langsmith Error: {response.status_code} - {response.text}")
|
||||
verbose_logger.error("Langsmith Error: %s - %s", response.status_code, response.text)
|
||||
else:
|
||||
if self.is_mock_mode:
|
||||
verbose_logger.debug(f"[LANGSMITH MOCK] Batch of {len(elements_to_log)} runs successfully mocked")
|
||||
verbose_logger.debug("[LANGSMITH MOCK] Batch of %s runs successfully mocked", len(elements_to_log))
|
||||
else:
|
||||
verbose_logger.debug(f"Batch of {len(self.log_queue)} runs successfully created")
|
||||
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
|
||||
except httpx.HTTPStatusError as e:
|
||||
verbose_logger.exception(f"Langsmith HTTP Error: {e.response.status_code} - {e.response.text}")
|
||||
verbose_logger.exception("Langsmith HTTP Error: %s - %s", e.response.status_code, e.response.text)
|
||||
except Exception:
|
||||
verbose_logger.exception(f"Langsmith Layer Error - {traceback.format_exc()}")
|
||||
verbose_logger.exception("Langsmith Layer Error - %s", traceback.format_exc())
|
||||
|
||||
def _group_batches_by_credentials(self) -> dict[CredentialsKey, BatchGroup]:
|
||||
"""Groups queue objects by credentials using a proper key structure"""
|
||||
|
|
|
|||
|
|
@ -94,9 +94,9 @@ class LiteralAILogger(CustomBatchLogger):
|
|||
)
|
||||
|
||||
if response.status_code >= 300:
|
||||
verbose_logger.error(f"Literal AI Error: {response.status_code} - {response.text}")
|
||||
verbose_logger.error("Literal AI Error: %s - %s", response.status_code, response.text)
|
||||
else:
|
||||
verbose_logger.debug(f"Batch of {len(self.log_queue)} runs successfully created")
|
||||
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
|
||||
except Exception:
|
||||
verbose_logger.exception("Literal AI Layer Error")
|
||||
|
||||
|
|
@ -152,11 +152,11 @@ class LiteralAILogger(CustomBatchLogger):
|
|||
headers=self.headers,
|
||||
)
|
||||
if response.status_code >= 300:
|
||||
verbose_logger.error(f"Literal AI Error: {response.status_code} - {response.text}")
|
||||
verbose_logger.error("Literal AI Error: %s - %s", response.status_code, response.text)
|
||||
else:
|
||||
verbose_logger.debug(f"Batch of {len(self.log_queue)} runs successfully created")
|
||||
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
|
||||
except httpx.HTTPStatusError as e:
|
||||
verbose_logger.exception(f"Literal AI HTTP Error: {e.response.status_code} - {e.response.text}")
|
||||
verbose_logger.exception("Literal AI HTTP Error: %s - %s", e.response.status_code, e.response.text)
|
||||
except Exception:
|
||||
verbose_logger.exception("Literal AI Layer Error")
|
||||
|
||||
|
|
|
|||
|
|
@ -35,7 +35,7 @@ class LogfireLogger:
|
|||
if logfire.DEFAULT_LOGFIRE_INSTANCE.config.send_to_logfire:
|
||||
logfire.configure(token=os.getenv("LOGFIRE_TOKEN"))
|
||||
except Exception as e:
|
||||
print_verbose(f"Got exception on init logfire client {e!s}")
|
||||
print_verbose(f"Got exception on init logfire client {e}")
|
||||
raise e
|
||||
|
||||
def _get_span_config(self, payload) -> SpanConfig:
|
||||
|
|
@ -90,7 +90,7 @@ class LogfireLogger:
|
|||
try:
|
||||
import logfire
|
||||
|
||||
verbose_logger.debug(f"logfire Logging - Enters logging function for model {kwargs}")
|
||||
verbose_logger.debug("logfire Logging - Enters logging function for model %s", kwargs)
|
||||
|
||||
if not response_obj:
|
||||
response_obj = {}
|
||||
|
|
@ -159,4 +159,4 @@ class LogfireLogger:
|
|||
|
||||
print_verbose(f"Logfire Layer Logging - final response object: {response_obj}")
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Logfire Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.debug("Logfire Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
|
|
|||
|
|
@ -99,7 +99,7 @@ class MlflowLogger(CustomLogger):
|
|||
)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"MLflow Logging Error - {e}", stack_info=True)
|
||||
verbose_logger.debug("MLflow Logging Error - %s", e, stack_info=True)
|
||||
|
||||
def _handle_stream_event(self, kwargs, response_obj, start_time, end_time):
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -144,7 +144,7 @@ def create_mock_client_factory(config: MockClientConfig):
|
|||
):
|
||||
"""Monkey-patched AsyncHTTPHandler.post that intercepts API calls."""
|
||||
if isinstance(url, str) and _is_mock_url(url):
|
||||
verbose_logger.info(f"[{config.name} MOCK] POST to {url}")
|
||||
verbose_logger.info("[%s MOCK] POST to %s", config.name, url)
|
||||
await asyncio.sleep(_MOCK_LATENCY_SECONDS)
|
||||
return MockResponse(
|
||||
status_code=config.default_status_code,
|
||||
|
|
@ -172,7 +172,7 @@ def create_mock_client_factory(config: MockClientConfig):
|
|||
def _mock_sync_client_post(self, url, **kwargs):
|
||||
"""Monkey-patched httpx.Client.post that intercepts API calls."""
|
||||
if _is_mock_url(url):
|
||||
verbose_logger.info(f"[{config.name} MOCK] POST to {url} (sync)")
|
||||
verbose_logger.info("[%s MOCK] POST to %s (sync)", config.name, url)
|
||||
return MockResponse(
|
||||
status_code=config.default_status_code,
|
||||
json_data=config.default_json_data,
|
||||
|
|
@ -198,7 +198,7 @@ def create_mock_client_factory(config: MockClientConfig):
|
|||
):
|
||||
"""Monkey-patched HTTPHandler.post that intercepts API calls."""
|
||||
if isinstance(url, str) and _is_mock_url(url):
|
||||
verbose_logger.info(f"[{config.name} MOCK] POST to {url}")
|
||||
verbose_logger.info("[%s MOCK] POST to %s", config.name, url)
|
||||
import time
|
||||
|
||||
time.sleep(_MOCK_LATENCY_SECONDS)
|
||||
|
|
@ -236,29 +236,29 @@ def create_mock_client_factory(config: MockClientConfig):
|
|||
if _mocks_initialized:
|
||||
return
|
||||
|
||||
verbose_logger.debug(f"[{config.name} MOCK] Initializing {config.name} mock client...")
|
||||
verbose_logger.debug("[%s MOCK] Initializing %s mock client...", config.name, config.name)
|
||||
|
||||
if config.patch_async_handler and _original_async_handler_post is None:
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
||||
|
||||
_original_async_handler_post = AsyncHTTPHandler.post
|
||||
AsyncHTTPHandler.post = _mock_async_handler_post # type: ignore
|
||||
verbose_logger.debug(f"[{config.name} MOCK] Patched AsyncHTTPHandler.post")
|
||||
verbose_logger.debug("[%s MOCK] Patched AsyncHTTPHandler.post", config.name)
|
||||
|
||||
if config.patch_sync_client and _original_sync_client_post is None:
|
||||
_original_sync_client_post = httpx.Client.post
|
||||
httpx.Client.post = _mock_sync_client_post # type: ignore
|
||||
verbose_logger.debug(f"[{config.name} MOCK] Patched httpx.Client.post")
|
||||
verbose_logger.debug("[%s MOCK] Patched httpx.Client.post", config.name)
|
||||
|
||||
if config.patch_http_handler and _original_http_handler_post is None:
|
||||
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
||||
|
||||
_original_http_handler_post = HTTPHandler.post
|
||||
HTTPHandler.post = _mock_http_handler_post # type: ignore
|
||||
verbose_logger.debug(f"[{config.name} MOCK] Patched HTTPHandler.post")
|
||||
verbose_logger.debug("[%s MOCK] Patched HTTPHandler.post", config.name)
|
||||
|
||||
verbose_logger.debug(f"[{config.name} MOCK] Mock latency set to {_MOCK_LATENCY_SECONDS * 1000:.0f}ms")
|
||||
verbose_logger.debug(f"[{config.name} MOCK] {config.name} mock client initialization complete")
|
||||
verbose_logger.debug("[%s MOCK] %s mock client initialization complete", config.name, config.name)
|
||||
|
||||
_mocks_initialized = True
|
||||
|
||||
|
|
@ -274,7 +274,7 @@ def create_mock_client_factory(config: MockClientConfig):
|
|||
result = bool(result) if result is not None else False
|
||||
|
||||
if result:
|
||||
verbose_logger.info(f"{config.name} Mock Mode: ENABLED - API calls will be mocked")
|
||||
verbose_logger.info("%s Mock Mode: ENABLED - API calls will be mocked", config.name)
|
||||
|
||||
return result
|
||||
|
||||
|
|
|
|||
|
|
@ -116,11 +116,12 @@ class NewRelicLogger(CustomLogger):
|
|||
|
||||
self.enabled = True
|
||||
verbose_logger.info(
|
||||
f"New Relic AI Monitoring initialized for app: {self.app_name}, "
|
||||
f"content recording: {self.record_content}"
|
||||
"New Relic AI Monitoring initialized for app: %s, content recording: %s",
|
||||
self.app_name,
|
||||
self.record_content,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Failed to initialize New Relic agent: {e}. Integration will be disabled.")
|
||||
verbose_logger.error("Failed to initialize New Relic agent: %s. Integration will be disabled.", e)
|
||||
self.enabled = False
|
||||
|
||||
def _get_newrelic_params(self) -> dict:
|
||||
|
|
@ -170,9 +171,10 @@ class NewRelicLogger(CustomLogger):
|
|||
if value in ("0", "false", "no", "off"):
|
||||
return False
|
||||
verbose_logger.warning(
|
||||
f"{var_name}={raw!r} is not a recognised boolean "
|
||||
f"(accepts true/false, 1/0, yes/no, on/off). "
|
||||
f"Falling back to default ({default})."
|
||||
"%s=%r is not a recognised boolean (accepts true/false, 1/0, yes/no, on/off). Falling back to default (%s).",
|
||||
var_name,
|
||||
raw,
|
||||
default,
|
||||
)
|
||||
return default
|
||||
|
||||
|
|
@ -188,7 +190,7 @@ class NewRelicLogger(CustomLogger):
|
|||
|
||||
return version("litellm")
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Unable to determine litellm version: {e}")
|
||||
verbose_logger.warning("Unable to determine litellm version: %s", e)
|
||||
return "unknown"
|
||||
|
||||
def _emit_supportability_metric(self):
|
||||
|
|
@ -216,12 +218,12 @@ class NewRelicLogger(CustomLogger):
|
|||
|
||||
if app and app.enabled:
|
||||
app.record_custom_metric(metric_name, 1)
|
||||
verbose_logger.info(f"Emitted New Relic supportability metric: {metric_name}")
|
||||
verbose_logger.info("Emitted New Relic supportability metric: %s", metric_name)
|
||||
else:
|
||||
verbose_logger.info("New Relic application is not enabled; skipping metric recording.")
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Failed to emit supportability metric: {e}")
|
||||
verbose_logger.warning("Failed to emit supportability metric: %s", e)
|
||||
|
||||
def _check_and_emit_periodic_metric(self):
|
||||
"""
|
||||
|
|
@ -294,14 +296,13 @@ class NewRelicLogger(CustomLogger):
|
|||
trace_id = slo_trace_id
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Unable to parse New Relic trace context from upstream sources: {e}")
|
||||
verbose_logger.warning("Unable to parse New Relic trace context from upstream sources: %s", e)
|
||||
|
||||
if not trace_id:
|
||||
trace_id = uuid.uuid4().hex
|
||||
verbose_logger.debug(
|
||||
f"New Relic trace_id not available from distributed tracing headers or "
|
||||
f"StandardLoggingPayload. Generated trace_id={trace_id} for AI monitoring "
|
||||
f"event grouping."
|
||||
"New Relic trace_id not available from distributed tracing headers or StandardLoggingPayload. Generated trace_id=%s for AI monitoring event grouping.",
|
||||
trace_id,
|
||||
)
|
||||
|
||||
return trace_id
|
||||
|
|
@ -638,7 +639,7 @@ class NewRelicLogger(CustomLogger):
|
|||
verbose_logger.warning("New Relic application is not enabled; skipping summary event recording.")
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Failed to record New Relic summary event: {e}")
|
||||
verbose_logger.warning("Failed to record New Relic summary event: %s", e)
|
||||
self.handle_callback_failure("newrelic")
|
||||
|
||||
def _record_message_events(
|
||||
|
|
@ -699,7 +700,7 @@ class NewRelicLogger(CustomLogger):
|
|||
app.record_custom_event("LlmChatCompletionMessage", event_data)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Failed to record New Relic message events: {e}")
|
||||
verbose_logger.warning("Failed to record New Relic message events: %s", e)
|
||||
self.handle_callback_failure("newrelic")
|
||||
|
||||
def _record_error_metric(self):
|
||||
|
|
@ -714,7 +715,7 @@ class NewRelicLogger(CustomLogger):
|
|||
if app and app.enabled:
|
||||
app.record_custom_metric("LLM/LiteLLM/Error", 1)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Failed to record New Relic error metric: {e}")
|
||||
verbose_logger.warning("Failed to record New Relic error metric: %s", e)
|
||||
self.handle_callback_failure("newrelic")
|
||||
|
||||
def _process_success(
|
||||
|
|
@ -846,7 +847,7 @@ class NewRelicLogger(CustomLogger):
|
|||
try:
|
||||
self._process_success(kwargs, response_obj, start_time, end_time)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error in New Relic log_success_event: {e}")
|
||||
verbose_logger.warning("Error in New Relic log_success_event: %s", e)
|
||||
self.handle_callback_failure("newrelic")
|
||||
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -859,7 +860,7 @@ class NewRelicLogger(CustomLogger):
|
|||
try:
|
||||
self._process_success(kwargs, response_obj, start_time, end_time)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error in New Relic async_log_success_event: {e}")
|
||||
verbose_logger.warning("Error in New Relic async_log_success_event: %s", e)
|
||||
self.handle_callback_failure("newrelic")
|
||||
|
||||
def log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -872,7 +873,7 @@ class NewRelicLogger(CustomLogger):
|
|||
self._record_error_metric()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error in New Relic log_failure_event: {e}")
|
||||
verbose_logger.warning("Error in New Relic log_failure_event: %s", e)
|
||||
self.handle_callback_failure("newrelic")
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -885,5 +886,5 @@ class NewRelicLogger(CustomLogger):
|
|||
self._record_error_metric()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error in New Relic async_log_failure_event: {e}")
|
||||
verbose_logger.warning("Error in New Relic async_log_failure_event: %s", e)
|
||||
self.handle_callback_failure("newrelic")
|
||||
|
|
|
|||
|
|
@ -24,6 +24,10 @@ from litellm.integrations.opentelemetry_utils.gen_ai_semconv import (
|
|||
from litellm.integrations.otel.model.semconv import Metric
|
||||
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
|
||||
from litellm.litellm_core_utils.secret_redaction import redact_string
|
||||
from litellm.litellm_core_utils.service_tier_utils import (
|
||||
get_requested_service_tier,
|
||||
get_served_service_tier,
|
||||
)
|
||||
from litellm.secret_managers.main import get_secret_bool, str_to_bool
|
||||
from litellm.types.services import ServiceLoggerPayload
|
||||
from litellm.types.utils import (
|
||||
|
|
@ -74,6 +78,11 @@ PREPROCESSING_DURATION_MS_ATTRIBUTE = "litellm.preprocessing.duration_ms"
|
|||
TEAM_METADATA_ATTRIBUTE = "litellm.team.metadata"
|
||||
MODEL_GROUP_ATTRIBUTE = "litellm.model_group"
|
||||
PROVIDER_MODEL_ATTRIBUTE = "litellm.provider.model"
|
||||
# semconv names the service tier attributes under the openai namespace, but every
|
||||
# provider that reports a tier (OpenAI, Anthropic, Bedrock, Groq, Vertex) uses the
|
||||
# same request param and response field, so both keys carry all of them.
|
||||
REQUEST_SERVICE_TIER_ATTRIBUTE = "gen_ai.openai.request.service_tier"
|
||||
RESPONSE_SERVICE_TIER_ATTRIBUTE = "gen_ai.openai.response.service_tier"
|
||||
# Remove the hardcoded LITELLM_RESOURCE dictionary - we'll create it properly later
|
||||
RAW_REQUEST_SPAN_NAME = "raw_gen_ai_request"
|
||||
LITELLM_REQUEST_SPAN_NAME = "litellm_request"
|
||||
|
|
@ -1411,6 +1420,23 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger):
|
|||
if provider_model:
|
||||
self.safe_set_attribute(span=span, key=PROVIDER_MODEL_ATTRIBUTE, value=provider_model)
|
||||
|
||||
def _set_service_tier_attributes(
|
||||
self,
|
||||
span: Span,
|
||||
standard_logging_payload: StandardLoggingPayload,
|
||||
) -> None:
|
||||
"""Stamp the tier the caller asked for and the tier the provider reports it
|
||||
served, so tier usage is segmentable in traces. Both are optional: a caller
|
||||
may not name a tier, and streaming responses carry no served tier.
|
||||
"""
|
||||
requested_tier = get_requested_service_tier(standard_logging_payload)
|
||||
if requested_tier is not None:
|
||||
self.safe_set_attribute(span=span, key=REQUEST_SERVICE_TIER_ATTRIBUTE, value=requested_tier)
|
||||
|
||||
served_tier = get_served_service_tier(standard_logging_payload)
|
||||
if served_tier is not None:
|
||||
self.safe_set_attribute(span=span, key=RESPONSE_SERVICE_TIER_ATTRIBUTE, value=served_tier)
|
||||
|
||||
@staticmethod
|
||||
def _team_metadata_json(value: Any, allowed_keys: list[str]) -> str | None:
|
||||
"""JSON-serialize only the allowlisted sub-keys of a team's metadata.
|
||||
|
|
@ -2310,6 +2336,8 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger):
|
|||
value=response_obj.get("model"),
|
||||
)
|
||||
|
||||
self._set_service_tier_attributes(span=span, standard_logging_payload=standard_logging_payload)
|
||||
|
||||
usage = response_obj and response_obj.get("usage")
|
||||
if usage:
|
||||
self.safe_set_attribute(
|
||||
|
|
@ -2688,7 +2716,8 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger):
|
|||
)
|
||||
except json.JSONDecodeError:
|
||||
verbose_logger.debug(
|
||||
f"litellm.integrations.opentelemetry.py::set_raw_request_attributes() - raw_response not json string - {_raw_response}"
|
||||
"litellm.integrations.opentelemetry.py::set_raw_request_attributes() - raw_response not json string - %s",
|
||||
_raw_response,
|
||||
)
|
||||
|
||||
self.safe_set_attribute(
|
||||
|
|
|
|||
|
|
@ -81,7 +81,7 @@ class OpikLogger(CustomBatchLogger):
|
|||
self.flush_lock: asyncio.Lock | None = asyncio.Lock()
|
||||
except Exception as e:
|
||||
verbose_logger.exception(
|
||||
f"OpikLogger - Asynchronous processing not initialized as we are not running in an async context {e!s}"
|
||||
"OpikLogger - Asynchronous processing not initialized as we are not running in an async context %s", e
|
||||
)
|
||||
self.flush_lock = None
|
||||
|
||||
|
|
@ -154,14 +154,14 @@ class OpikLogger(CustomBatchLogger):
|
|||
self.log_queue.append(span_payload.__dict__)
|
||||
|
||||
verbose_logger.debug(
|
||||
f"OpikLogger added event to log_queue - Will flush in {self.flush_interval} seconds..."
|
||||
"OpikLogger added event to log_queue - Will flush in %s seconds...", self.flush_interval
|
||||
)
|
||||
|
||||
if len(self.log_queue) >= self.batch_size:
|
||||
verbose_logger.debug("OpikLogger - Flushing batch")
|
||||
await self.flush_queue()
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"OpikLogger failed to log success event - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("OpikLogger failed to log success event - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
def _sync_send(self, url: str, headers: dict[str, str], batch: dict[str, Any]) -> None:
|
||||
try:
|
||||
|
|
@ -174,7 +174,7 @@ class OpikLogger(CustomBatchLogger):
|
|||
if response.status_code != 204:
|
||||
raise Exception(f"Response from opik API status_code: {response.status_code}, text: {response.text}")
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"OpikLogger failed to send batch - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("OpikLogger failed to send batch - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
def log_success_event(
|
||||
self,
|
||||
|
|
@ -245,7 +245,7 @@ class OpikLogger(CustomBatchLogger):
|
|||
batch={"spans": [span_payload.__dict__]},
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"OpikLogger failed to log success event - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("OpikLogger failed to log success event - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def _submit_batch(self, url: str, headers: dict[str, str], batch: dict[str, Any]) -> None:
|
||||
try:
|
||||
|
|
@ -257,11 +257,11 @@ class OpikLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
|
||||
if response.status_code >= 300:
|
||||
verbose_logger.error(f"OpikLogger - Error: {response.status_code} - {response.text}")
|
||||
verbose_logger.error("OpikLogger - Error: %s - %s", response.status_code, response.text)
|
||||
else:
|
||||
verbose_logger.info(f"OpikLogger - {len(self.log_queue)} Opik events submitted")
|
||||
verbose_logger.info("OpikLogger - %s Opik events submitted", len(self.log_queue))
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"OpikLogger failed to send batch - {e!s}")
|
||||
verbose_logger.exception("OpikLogger failed to send batch - %s", e)
|
||||
|
||||
def _create_opik_headers(self) -> dict[str, str]:
|
||||
headers: dict[str, str] = {}
|
||||
|
|
@ -283,7 +283,7 @@ class OpikLogger(CustomBatchLogger):
|
|||
# Send trace batch
|
||||
if len(traces) > 0:
|
||||
await self._submit_batch(url=self.trace_url, headers=self.headers, batch={"traces": traces})
|
||||
verbose_logger.info(f"Sent {len(traces)} traces")
|
||||
verbose_logger.info("Sent %s traces", len(traces))
|
||||
if len(spans) > 0:
|
||||
await self._submit_batch(url=self.span_url, headers=self.headers, batch={"spans": spans})
|
||||
verbose_logger.info(f"Sent {len(spans)} spans")
|
||||
verbose_logger.info("Sent %s spans", len(spans))
|
||||
|
|
|
|||
|
|
@ -66,7 +66,7 @@ def extract_opik_metadata(
|
|||
if requester_opik:
|
||||
opik_meta.update(requester_opik)
|
||||
|
||||
_logging.verbose_logger.debug(f"litellm_opik_metadata - {json.dumps(opik_meta, default=str)}")
|
||||
_logging.verbose_logger.debug("litellm_opik_metadata - %s", json.dumps(opik_meta, default=str))
|
||||
|
||||
return opik_meta
|
||||
|
||||
|
|
@ -92,7 +92,7 @@ def extract_span_identifiers(
|
|||
try:
|
||||
return current_span_data.trace_id, current_span_data.id
|
||||
except AttributeError:
|
||||
_logging.verbose_logger.warning(f"Unexpected current_span_data format: {type(current_span_data)}")
|
||||
_logging.verbose_logger.warning("Unexpected current_span_data format: %s", type(current_span_data))
|
||||
return None, None
|
||||
|
||||
|
||||
|
|
@ -152,7 +152,7 @@ def apply_proxy_header_overrides(
|
|||
if isinstance(parsed_tags, list):
|
||||
tags.extend(parsed_tags)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
_logging.verbose_logger.warning(f"Failed to parse tags from header: {value}")
|
||||
_logging.verbose_logger.warning("Failed to parse tags from header: %s", value)
|
||||
|
||||
return project_name, tags, thread_id
|
||||
|
||||
|
|
|
|||
|
|
@ -61,7 +61,7 @@ def build_span_payload(
|
|||
created = response_obj.get("created", 0)
|
||||
span_name = f"{model}_{obj_type}_{created}"
|
||||
|
||||
_logging.verbose_logger.debug(f"OpikLogger creating span with id {span_id} for trace {trace_id}")
|
||||
_logging.verbose_logger.debug("OpikLogger creating span with id %s for trace %s", span_id, trace_id)
|
||||
|
||||
return types.SpanPayload(
|
||||
id=span_id,
|
||||
|
|
|
|||
|
|
@ -72,7 +72,7 @@ class PostHogLogger(CustomBatchLogger):
|
|||
super().__init__(**kwargs, flush_lock=None, batch_size=POSTHOG_MAX_BATCH_SIZE)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"PostHog: Got exception on init PostHog client {e!s}")
|
||||
verbose_logger.exception("PostHog: Got exception on init PostHog client %s", e)
|
||||
raise e
|
||||
|
||||
def log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
|
|
@ -107,7 +107,7 @@ class PostHogLogger(CustomBatchLogger):
|
|||
verbose_logger.debug("PostHog: Sync event successfully sent")
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"PostHog Sync Layer Error - {e!s}")
|
||||
verbose_logger.exception("PostHog Sync Layer Error - %s", e)
|
||||
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
|
|
@ -115,7 +115,7 @@ class PostHogLogger(CustomBatchLogger):
|
|||
self._ensure_async_setup() # Lazy initialization
|
||||
await self._log_async_event(kwargs, response_obj, start_time, end_time)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"PostHog Layer Error - {e!s}")
|
||||
verbose_logger.exception("PostHog Layer Error - %s", e)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
|
|
@ -123,7 +123,7 @@ class PostHogLogger(CustomBatchLogger):
|
|||
self._ensure_async_setup() # Lazy initialization
|
||||
await self._log_async_event(kwargs, response_obj, start_time, end_time)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"PostHog Layer Error - {e!s}")
|
||||
verbose_logger.exception("PostHog Layer Error - %s", e)
|
||||
|
||||
async def _log_async_event(self, kwargs, response_obj=None, start_time=0.0, end_time=0.0):
|
||||
# Note: response_obj, start_time, end_time not used - all data comes from kwargs
|
||||
|
|
@ -132,7 +132,7 @@ class PostHogLogger(CustomBatchLogger):
|
|||
|
||||
# Store event with its credentials for batch sending
|
||||
self.log_queue.append({"event": event_payload, "api_key": api_key, "api_url": api_url})
|
||||
verbose_logger.debug(f"PostHog, event added to queue. Will flush in {self.flush_interval} seconds...")
|
||||
verbose_logger.debug("PostHog, event added to queue. Will flush in %s seconds...", self.flush_interval)
|
||||
|
||||
if len(self.log_queue) >= self.batch_size:
|
||||
await self.flush_queue()
|
||||
|
|
@ -328,7 +328,7 @@ class PostHogLogger(CustomBatchLogger):
|
|||
if not self.log_queue:
|
||||
return
|
||||
|
||||
verbose_logger.debug(f"PostHog: Sending batch of {len(self.log_queue)} events")
|
||||
verbose_logger.debug("PostHog: Sending batch of %s events", len(self.log_queue))
|
||||
|
||||
if self.is_mock_mode:
|
||||
verbose_logger.debug("[POSTHOG MOCK] Mock mode enabled - API calls will be intercepted")
|
||||
|
|
@ -363,11 +363,11 @@ class PostHogLogger(CustomBatchLogger):
|
|||
)
|
||||
|
||||
if self.is_mock_mode:
|
||||
verbose_logger.debug(f"[POSTHOG MOCK] Batch of {len(self.log_queue)} events successfully mocked")
|
||||
verbose_logger.debug("[POSTHOG MOCK] Batch of %s events successfully mocked", len(self.log_queue))
|
||||
else:
|
||||
verbose_logger.debug(f"PostHog: Batch of {len(self.log_queue)} events successfully sent")
|
||||
verbose_logger.debug("PostHog: Batch of %s events successfully sent", len(self.log_queue))
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"PostHog Error sending batch API - {e!s}")
|
||||
verbose_logger.exception("PostHog Error sending batch API - %s", e)
|
||||
|
||||
def _ensure_async_setup(self):
|
||||
if not self._async_initialized:
|
||||
|
|
@ -377,7 +377,7 @@ class PostHogLogger(CustomBatchLogger):
|
|||
self._async_initialized = True
|
||||
verbose_logger.debug("PostHog: Async components initialized")
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"PostHog: Failed to initialize async components: {e!s}")
|
||||
verbose_logger.error("PostHog: Failed to initialize async components: %s", e)
|
||||
raise
|
||||
|
||||
def _extract_metadata(self, kwargs: dict[str, Any]) -> dict[str, Any]:
|
||||
|
|
@ -408,7 +408,7 @@ class PostHogLogger(CustomBatchLogger):
|
|||
if not self.log_queue:
|
||||
return
|
||||
|
||||
verbose_logger.debug(f"PostHog: Flushing {len(self.log_queue)} remaining events on exit")
|
||||
verbose_logger.debug("PostHog: Flushing %s remaining events on exit", len(self.log_queue))
|
||||
|
||||
try:
|
||||
# Group events by credentials (same logic as async_send_batch)
|
||||
|
|
@ -436,13 +436,13 @@ class PostHogLogger(CustomBatchLogger):
|
|||
response.raise_for_status()
|
||||
|
||||
if response.status_code != 200:
|
||||
verbose_logger.error(f"PostHog: Failed to flush on exit - status {response.status_code}")
|
||||
verbose_logger.error("PostHog: Failed to flush on exit - status %s", response.status_code)
|
||||
|
||||
if self.is_mock_mode:
|
||||
verbose_logger.debug(f"[POSTHOG MOCK] Successfully flushed {len(self.log_queue)} events on exit")
|
||||
verbose_logger.debug("[POSTHOG MOCK] Successfully flushed %s events on exit", len(self.log_queue))
|
||||
else:
|
||||
verbose_logger.debug(f"PostHog: Successfully flushed {len(self.log_queue)} events on exit")
|
||||
verbose_logger.debug("PostHog: Successfully flushed %s events on exit", len(self.log_queue))
|
||||
self.log_queue.clear()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"PostHog: Error flushing events on exit: {e!s}")
|
||||
verbose_logger.error("PostHog: Error flushing events on exit: %s", e)
|
||||
|
|
|
|||
|
|
@ -34,6 +34,9 @@ from litellm.litellm_core_utils.core_helpers import (
|
|||
get_litellm_metadata_from_kwargs,
|
||||
get_metadata_variable_name_from_kwargs,
|
||||
)
|
||||
from litellm.litellm_core_utils.service_tier_utils import (
|
||||
get_service_tier_from_standard_logging_payload,
|
||||
)
|
||||
from litellm.proxy._types import (
|
||||
LiteLLM_DeletedVerificationToken,
|
||||
LiteLLM_TeamTable,
|
||||
|
|
@ -98,16 +101,6 @@ class _ExcludedLabelMetric:
|
|||
return self._metric.labels(*kept_values) if kept_values else self._metric
|
||||
|
||||
|
||||
# Tiers a caller may name in a request, across the providers that accept the
|
||||
# parameter: OpenAI ("auto", "default", "flex", "priority", "scale"), Bedrock and
|
||||
# Groq (subsets of those), Anthropic ("auto", "standard_only") and Vertex, which
|
||||
# maps "default" to "standard". Used to bound the caller-controlled fallback in
|
||||
# ``get_service_tier_from_standard_logging_payload``.
|
||||
KNOWN_REQUEST_SERVICE_TIERS = frozenset(
|
||||
{"auto", "batch", "default", "flex", "priority", "scale", "standard", "standard_only"}
|
||||
)
|
||||
|
||||
|
||||
def _get_budget_metrics_per_request_timeout() -> float:
|
||||
raw = os.getenv("PROMETHEUS_BUDGET_METRICS_PER_REQUEST_TIMEOUT")
|
||||
if raw is None:
|
||||
|
|
@ -683,7 +676,7 @@ class PrometheusLogger(CustomLogger):
|
|||
)
|
||||
|
||||
except Exception as e:
|
||||
print_verbose(f"Got exception on init prometheus client {e!s}")
|
||||
print_verbose(f"Got exception on init prometheus client {e}")
|
||||
raise e
|
||||
|
||||
def _parse_prometheus_config(self) -> dict[str, list[str]]:
|
||||
|
|
@ -697,7 +690,7 @@ class PrometheusLogger(CustomLogger):
|
|||
if not config:
|
||||
return {}
|
||||
|
||||
verbose_logger.debug(f"prometheus config: {config}")
|
||||
verbose_logger.debug("prometheus config: %s", config)
|
||||
|
||||
# Parse and validate all configuration groups
|
||||
parsed_configs = []
|
||||
|
|
@ -963,7 +956,10 @@ class PrometheusLogger(CustomLogger):
|
|||
except ImportError:
|
||||
# Fallback to simple logging if rich is not available
|
||||
verbose_logger.error(
|
||||
f"Invalid labels for metric '{metric_name}': {invalid_labels}. Valid labels: {sorted(valid_labels)}"
|
||||
"Invalid labels for metric '%s': %s. Valid labels: %s",
|
||||
metric_name,
|
||||
invalid_labels,
|
||||
sorted(valid_labels),
|
||||
)
|
||||
|
||||
def _pretty_print_invalid_metric_error(self, invalid_metric_name: str, valid_metrics: tuple) -> None:
|
||||
|
|
@ -1003,7 +999,9 @@ class PrometheusLogger(CustomLogger):
|
|||
|
||||
except ImportError:
|
||||
# Fallback to simple logging if rich is not available
|
||||
verbose_logger.error(f"Invalid metric name: {invalid_metric_name}. Valid metrics: {sorted(valid_metrics)}")
|
||||
verbose_logger.error(
|
||||
"Invalid metric name: %s. Valid metrics: %s", invalid_metric_name, sorted(valid_metrics)
|
||||
)
|
||||
|
||||
#########################################################
|
||||
# End of pretty print functions
|
||||
|
|
@ -1078,9 +1076,10 @@ class PrometheusLogger(CustomLogger):
|
|||
except ImportError:
|
||||
# Fallback to simple logging if rich is not available
|
||||
verbose_logger.info(
|
||||
f"Enabled metrics: {sorted(self.enabled_metrics) if hasattr(self, 'enabled_metrics') else 'All metrics'}"
|
||||
"Enabled metrics: %s",
|
||||
sorted(self.enabled_metrics) if hasattr(self, "enabled_metrics") else "All metrics",
|
||||
)
|
||||
verbose_logger.info(f"Label filters: {label_filters}")
|
||||
verbose_logger.info("Label filters: %s", label_filters)
|
||||
|
||||
def _is_metric_enabled(self, metric_name: str) -> bool:
|
||||
"""Check if a metric is enabled based on configuration"""
|
||||
|
|
@ -1866,7 +1865,9 @@ class PrometheusLogger(CustomLogger):
|
|||
for i, r in enumerate(results):
|
||||
if isinstance(r, Exception):
|
||||
verbose_logger.debug(
|
||||
f"[Non-Blocking] Prometheus: Budget metric lookup {['key', 'team', 'user', 'org'][i]} failed: {r}"
|
||||
"[Non-Blocking] Prometheus: Budget metric lookup %s failed: %s",
|
||||
["key", "team", "user", "org"][i],
|
||||
r,
|
||||
)
|
||||
|
||||
def _increment_top_level_request_and_spend_metrics(
|
||||
|
|
@ -2132,7 +2133,7 @@ class PrometheusLogger(CustomLogger):
|
|||
response_cost=0,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"prometheus Layer Error(): Exception occured - {e!s}")
|
||||
verbose_logger.exception("prometheus Layer Error(): Exception occured - %s", e)
|
||||
|
||||
def _extract_status_code(
|
||||
self,
|
||||
|
|
@ -2262,8 +2263,9 @@ class PrometheusLogger(CustomLogger):
|
|||
|
||||
if self._is_invalid_api_key_request(status_code, exception=exception):
|
||||
verbose_logger.debug(
|
||||
"Skipping Prometheus metrics for invalid API key request: "
|
||||
f"status_code={status_code}, exception={type(exception).__name__ if exception else None}"
|
||||
"Skipping Prometheus metrics for invalid API key request: status_code=%s, exception=%s",
|
||||
status_code,
|
||||
type(exception).__name__ if exception else None,
|
||||
)
|
||||
return True
|
||||
|
||||
|
|
@ -2383,7 +2385,7 @@ class PrometheusLogger(CustomLogger):
|
|||
)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"prometheus Layer Error(): Exception occured - {e!s}")
|
||||
verbose_logger.exception("prometheus Layer Error(): Exception occured - %s", e)
|
||||
|
||||
async def async_post_call_success_hook(self, data: dict, user_api_key_dict: UserAPIKeyAuth, response):
|
||||
"""
|
||||
|
|
@ -2608,7 +2610,7 @@ class PrometheusLogger(CustomLogger):
|
|||
)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Prometheus Error: set_llm_deployment_failure_metrics. Exception occured - {e!s}")
|
||||
verbose_logger.debug("Prometheus Error: set_llm_deployment_failure_metrics. Exception occured - %s", e)
|
||||
|
||||
def _set_deployment_tpm_rpm_limit_metrics(
|
||||
self,
|
||||
|
|
@ -2722,9 +2724,7 @@ class PrometheusLogger(CustomLogger):
|
|||
)
|
||||
self.litellm_remaining_requests_metric.labels(**_labels).set(remaining_requests)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(
|
||||
f"Prometheus Error: _async_set_router_remaining_metrics. Exception occured - {e!s}"
|
||||
)
|
||||
verbose_logger.exception("Prometheus Error: _async_set_router_remaining_metrics. Exception occured - %s", e)
|
||||
|
||||
def set_llm_deployment_success_metrics(
|
||||
self,
|
||||
|
|
@ -2867,7 +2867,7 @@ class PrometheusLogger(CustomLogger):
|
|||
self.litellm_deployment_latency_per_output_token.labels(**_labels).observe(latency_per_token)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Prometheus Error: set_llm_deployment_success_metrics. Exception occured - {e!s}")
|
||||
verbose_logger.exception("Prometheus Error: set_llm_deployment_success_metrics. Exception occured - %s", e)
|
||||
return
|
||||
|
||||
def _record_guardrail_metrics(
|
||||
|
|
@ -2912,7 +2912,7 @@ class PrometheusLogger(CustomLogger):
|
|||
hook_type=hook_type,
|
||||
).inc()
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Error recording guardrail metrics: {e!s}")
|
||||
verbose_logger.debug("Error recording guardrail metrics: %s", e)
|
||||
|
||||
########################################
|
||||
# Managed Batch Metric Recording Methods
|
||||
|
|
@ -2935,7 +2935,7 @@ class PrometheusLogger(CustomLogger):
|
|||
api_key_alias=api_key_alias,
|
||||
).inc()
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error recording batch created metric: {e}")
|
||||
verbose_logger.warning("Error recording batch created metric: %s", e)
|
||||
|
||||
def record_managed_file_size(
|
||||
self,
|
||||
|
|
@ -2956,7 +2956,7 @@ class PrometheusLogger(CustomLogger):
|
|||
user=user or "",
|
||||
).set(size_bytes)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error recording file size metric: {e}")
|
||||
verbose_logger.warning("Error recording file size metric: %s", e)
|
||||
|
||||
def record_managed_batch_duration(
|
||||
self,
|
||||
|
|
@ -2970,7 +2970,7 @@ class PrometheusLogger(CustomLogger):
|
|||
api_provider=api_provider or "",
|
||||
).observe(duration_seconds)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error recording batch duration metric: {e}")
|
||||
verbose_logger.warning("Error recording batch duration metric: %s", e)
|
||||
|
||||
def record_managed_file_created(
|
||||
self,
|
||||
|
|
@ -2989,14 +2989,14 @@ class PrometheusLogger(CustomLogger):
|
|||
api_key_alias=api_key_alias,
|
||||
).inc()
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error recording file created metric: {e}")
|
||||
verbose_logger.warning("Error recording file created metric: %s", e)
|
||||
|
||||
def record_managed_file_deleted(self, result: str):
|
||||
"""Record a managed file deletion attempt. result is 'success' or 'blocked'."""
|
||||
try:
|
||||
self.litellm_managed_file_deleted_total.labels(result=result).inc()
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error recording file deleted metric: {e}")
|
||||
verbose_logger.warning("Error recording file deleted metric: %s", e)
|
||||
|
||||
def record_check_batch_cost_run(
|
||||
self,
|
||||
|
|
@ -3023,7 +3023,7 @@ class PrometheusLogger(CustomLogger):
|
|||
api_provider=api_provider or "",
|
||||
).inc()
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error recording check batch cost metrics: {e}")
|
||||
verbose_logger.warning("Error recording check batch cost metrics: %s", e)
|
||||
|
||||
def record_check_batch_cost_error(self, error_type: str):
|
||||
try:
|
||||
|
|
@ -3031,7 +3031,7 @@ class PrometheusLogger(CustomLogger):
|
|||
error_type=error_type,
|
||||
).inc()
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error recording check batch cost error metric: {e}")
|
||||
verbose_logger.warning("Error recording check batch cost error metric: %s", e)
|
||||
|
||||
@staticmethod
|
||||
def _get_exception_class_name(exception: Exception) -> str:
|
||||
|
|
@ -3315,7 +3315,7 @@ class PrometheusLogger(CustomLogger):
|
|||
await set_metrics_function(data)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error initializing {data_type} budget metrics: {e!s}")
|
||||
verbose_logger.exception("Error initializing %s budget metrics: %s", data_type, e)
|
||||
|
||||
async def _initialize_team_budget_metrics(self):
|
||||
"""
|
||||
|
|
@ -3495,18 +3495,18 @@ class PrometheusLogger(CustomLogger):
|
|||
# Get total user count
|
||||
total_users = await UserRepository(prisma_client).table.count()
|
||||
self.litellm_total_users_metric.set(total_users)
|
||||
verbose_logger.debug(f"Prometheus: set litellm_total_users to {total_users}")
|
||||
verbose_logger.debug("Prometheus: set litellm_total_users to %s", total_users)
|
||||
|
||||
billable_users = await UserRepository(prisma_client).count_billable_users()
|
||||
self.litellm_active_users_metric.set(billable_users)
|
||||
verbose_logger.debug(f"Prometheus: set litellm_active_users to {billable_users}")
|
||||
verbose_logger.debug("Prometheus: set litellm_active_users to %s", billable_users)
|
||||
|
||||
# Get total team count
|
||||
total_teams = await TeamRepository(prisma_client).table.count()
|
||||
self.litellm_teams_count_metric.set(total_teams)
|
||||
verbose_logger.debug(f"Prometheus: set litellm_teams_count to {total_teams}")
|
||||
verbose_logger.debug("Prometheus: set litellm_teams_count to %s", total_teams)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error initializing user/team count metrics: {e!s}")
|
||||
verbose_logger.exception("Error initializing user/team count metrics: %s", e)
|
||||
|
||||
async def _set_key_list_budget_metrics(self, keys: list[str | UserAPIKeyAuth]):
|
||||
"""Helper function to set budget metrics for a list of keys"""
|
||||
|
|
@ -3597,7 +3597,7 @@ class PrometheusLogger(CustomLogger):
|
|||
user_api_key_cache=user_api_key_cache,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"[Non-Blocking] Prometheus: Error getting team info: {e!s}")
|
||||
verbose_logger.debug("[Non-Blocking] Prometheus: Error getting team info: %s", e)
|
||||
return team_object
|
||||
|
||||
if team_info:
|
||||
|
|
@ -3695,7 +3695,7 @@ class PrometheusLogger(CustomLogger):
|
|||
include_budget_table=True,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"[Non-Blocking] Prometheus: Error getting org info: {e!s}")
|
||||
verbose_logger.debug("[Non-Blocking] Prometheus: Error getting org info: %s", e)
|
||||
return
|
||||
|
||||
if org_info is None:
|
||||
|
|
@ -3852,7 +3852,7 @@ class PrometheusLogger(CustomLogger):
|
|||
if key_object:
|
||||
user_api_key_dict.budget_reset_at = key_object.budget_reset_at
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"[Non-Blocking] Prometheus: Error getting key info: {e!s}")
|
||||
verbose_logger.debug("[Non-Blocking] Prometheus: Error getting key info: %s", e)
|
||||
|
||||
return user_api_key_dict
|
||||
|
||||
|
|
@ -3917,7 +3917,7 @@ class PrometheusLogger(CustomLogger):
|
|||
check_db_only=False,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"[Non-Blocking] Prometheus: Error getting user info: {e!s}")
|
||||
verbose_logger.debug("[Non-Blocking] Prometheus: Error getting user info: %s", e)
|
||||
return user_object
|
||||
|
||||
if user_info:
|
||||
|
|
@ -4174,44 +4174,6 @@ def get_custom_labels_from_metadata(metadata: dict) -> dict[str, str]:
|
|||
return result
|
||||
|
||||
|
||||
def get_service_tier_from_standard_logging_payload(
|
||||
standard_logging_payload: StandardLoggingPayload,
|
||||
) -> str | None:
|
||||
"""
|
||||
Resolve the service tier a request ran on, for the ``service_tier`` label.
|
||||
|
||||
The tier the provider actually served wins over the tier the caller asked for,
|
||||
so latency and spend stay segmentable when the request said ``auto`` and the
|
||||
provider picked the concrete tier. Providers report the served tier either at
|
||||
the top level of the response (OpenAI, Bedrock, Groq) or on the usage object
|
||||
(Anthropic).
|
||||
|
||||
Streaming responses carry no served tier, so the requested tier is the
|
||||
fallback. That value is caller-controlled and survives param mapping even
|
||||
where the provider then ignores it (Bedrock and Groq accept the request and
|
||||
drop an unrecognized tier), so it is only labelled when it names a known
|
||||
tier; otherwise one caller could mint a Prometheus series per string. Values
|
||||
the provider itself reports are not caller-controlled and stay unrestricted,
|
||||
so a tier a provider adds later is still labelled correctly.
|
||||
"""
|
||||
response = standard_logging_payload.get("response")
|
||||
usage_object = standard_logging_payload.get("metadata", {}).get("usage_object")
|
||||
|
||||
served_candidates: tuple[object, ...] = (
|
||||
response.get("service_tier") if isinstance(response, dict) else None,
|
||||
usage_object.get("service_tier") if isinstance(usage_object, dict) else None,
|
||||
)
|
||||
served_tier = next((tier for tier in served_candidates if isinstance(tier, str) and tier), None)
|
||||
if served_tier is not None:
|
||||
return served_tier
|
||||
|
||||
model_parameters = standard_logging_payload.get("model_parameters")
|
||||
requested_tier = model_parameters.get("service_tier") if isinstance(model_parameters, dict) else None
|
||||
if isinstance(requested_tier, str) and requested_tier in KNOWN_REQUEST_SERVICE_TIERS:
|
||||
return requested_tier
|
||||
return None
|
||||
|
||||
|
||||
def _get_combined_custom_metadata_from_standard_logging_payload(
|
||||
standard_logging_payload: dict | None,
|
||||
) -> dict[str, Any]:
|
||||
|
|
|
|||
|
|
@ -82,7 +82,7 @@ class PrometheusServicesLogger:
|
|||
self.mock_testing_failure_calls = 0
|
||||
|
||||
except Exception as e:
|
||||
print_verbose(f"Got exception on init prometheus client {e!s}")
|
||||
print_verbose(f"Got exception on init prometheus client {e}")
|
||||
raise e
|
||||
|
||||
def _get_service_metrics_initialize(self, service: ServiceTypes) -> list[ServiceMetrics]:
|
||||
|
|
@ -92,7 +92,7 @@ class PrometheusServicesLogger:
|
|||
|
||||
metrics = DEFAULT_SERVICE_CONFIGS.get(service, {}).get("metrics", [])
|
||||
if not metrics:
|
||||
verbose_logger.debug(f"No metrics found for service {service}")
|
||||
verbose_logger.debug("No metrics found for service %s", service)
|
||||
return DEFAULT_METRICS
|
||||
return metrics
|
||||
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -31,7 +31,7 @@ class S3Logger:
|
|||
import boto3
|
||||
|
||||
try:
|
||||
verbose_logger.debug(f"in init s3 logger - s3_callback_params {litellm.s3_callback_params}")
|
||||
verbose_logger.debug("in init s3 logger - s3_callback_params %s", litellm.s3_callback_params)
|
||||
|
||||
s3_use_team_prefix = False
|
||||
|
||||
|
|
@ -62,7 +62,7 @@ class S3Logger:
|
|||
self.s3_server_side_encryption, self.s3_sse_kms_key_id = resolve_sse_params(
|
||||
s3_server_side_encryption, s3_sse_kms_key_id
|
||||
)
|
||||
verbose_logger.debug(f"s3 logger using endpoint url {s3_endpoint_url}")
|
||||
verbose_logger.debug("s3 logger using endpoint url %s", s3_endpoint_url)
|
||||
# Create an S3 client with custom endpoint URL
|
||||
self.s3_client = boto3.client(
|
||||
"s3",
|
||||
|
|
@ -78,7 +78,7 @@ class S3Logger:
|
|||
**kwargs,
|
||||
)
|
||||
except Exception as e:
|
||||
print_verbose(f"Got exception on init s3 client {e!s}")
|
||||
print_verbose(f"Got exception on init s3 client {e}")
|
||||
raise e
|
||||
|
||||
async def _async_log_event(self, kwargs, response_obj, start_time, end_time, print_verbose):
|
||||
|
|
@ -86,7 +86,7 @@ class S3Logger:
|
|||
|
||||
def log_event(self, kwargs, response_obj, start_time, end_time, print_verbose):
|
||||
try:
|
||||
verbose_logger.debug(f"s3 Logging - Enters logging function for model {kwargs}")
|
||||
verbose_logger.debug("s3 Logging - Enters logging function for model %s", kwargs)
|
||||
|
||||
# construct payload to send to s3
|
||||
# follows the same params as langfuse.py
|
||||
|
|
@ -163,19 +163,19 @@ class S3Logger:
|
|||
**sse_params,
|
||||
)
|
||||
|
||||
print_verbose(f"Response from s3:{response!s}")
|
||||
print_verbose(f"Response from s3:{response}")
|
||||
|
||||
print_verbose(f"s3 Layer Logging - final response object: {response_obj}")
|
||||
return response
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"s3 Layer Error - {e!s}")
|
||||
verbose_logger.exception("s3 Layer Error - %s", e)
|
||||
|
||||
|
||||
def _validated_sse_value(name: str, value: str | None) -> str | None:
|
||||
if value is None or isinstance(value, str):
|
||||
return value
|
||||
verbose_logger.warning(
|
||||
f"s3 logging: ignoring {name} because it has invalid type {type(value).__name__}; expected a string"
|
||||
"s3 logging: ignoring %s because it has invalid type %s; expected a string", name, type(value).__name__
|
||||
)
|
||||
return None
|
||||
|
||||
|
|
@ -191,8 +191,8 @@ def resolve_sse_params(
|
|||
return None, None
|
||||
if valid_key_id and not algorithm.startswith("aws:kms"):
|
||||
verbose_logger.warning(
|
||||
f"s3 logging: ignoring s3_sse_kms_key_id because s3_server_side_encryption is {algorithm}; "
|
||||
"set it to aws:kms to encrypt with the KMS key"
|
||||
"s3 logging: ignoring s3_sse_kms_key_id because s3_server_side_encryption is %s; set it to aws:kms to encrypt with the KMS key",
|
||||
algorithm,
|
||||
)
|
||||
return algorithm, None
|
||||
return algorithm, valid_key_id
|
||||
|
|
|
|||
|
|
@ -64,12 +64,12 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
_masker = SensitiveDataMasker()
|
||||
if s3_callback_params_override is not None:
|
||||
verbose_logger.debug(
|
||||
f"in init s3 logger (audit override) - {_masker.mask_dict(dict(s3_callback_params_override))}"
|
||||
"in init s3 logger (audit override) - %s", _masker.mask_dict(dict(s3_callback_params_override))
|
||||
)
|
||||
else:
|
||||
verbose_logger.debug(
|
||||
f"in init s3 logger - s3_callback_params "
|
||||
f"{_masker.mask_dict(dict(litellm.s3_callback_params or {}))}"
|
||||
"in init s3 logger - s3_callback_params %s",
|
||||
_masker.mask_dict(dict(litellm.s3_callback_params or {})),
|
||||
)
|
||||
|
||||
# Initialize S3 params first to get the correct s3_verify value
|
||||
|
|
@ -98,11 +98,11 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
s3_server_side_encryption=s3_server_side_encryption,
|
||||
s3_sse_kms_key_id=s3_sse_kms_key_id,
|
||||
)
|
||||
verbose_logger.debug(f"s3 logger using endpoint url {s3_endpoint_url}")
|
||||
verbose_logger.debug("s3 logger using endpoint url %s", s3_endpoint_url)
|
||||
|
||||
# IMPORTANT
|
||||
# Create httpx client AFTER _init_s3_params so we have the correct s3_verify value
|
||||
verbose_logger.debug(f"s3_v2 logger creating async httpx client with s3_verify={self.s3_verify}")
|
||||
verbose_logger.debug("s3_v2 logger creating async httpx client with s3_verify=%s", self.s3_verify)
|
||||
self.async_httpx_client = get_async_httpx_client(
|
||||
llm_provider=httpxSpecialProvider.LoggingCallback,
|
||||
params={"ssl_verify": self.s3_verify},
|
||||
|
|
@ -111,7 +111,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
asyncio.create_task(self.periodic_flush())
|
||||
self.flush_lock = asyncio.Lock()
|
||||
|
||||
verbose_logger.debug(f"s3 flush interval: {s3_flush_interval}, s3 batch size: {s3_batch_size}")
|
||||
verbose_logger.debug("s3 flush interval: %s, s3 batch size: %s", s3_flush_interval, s3_batch_size)
|
||||
# Call CustomLogger's __init__
|
||||
CustomBatchLogger.__init__(
|
||||
self,
|
||||
|
|
@ -125,7 +125,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
BaseAWSLLM.__init__(self)
|
||||
|
||||
except Exception as e:
|
||||
print_verbose(f"Got exception on init s3 client {e!s}")
|
||||
print_verbose(f"Got exception on init s3 client {e}")
|
||||
raise e
|
||||
|
||||
def _init_s3_params(
|
||||
|
|
@ -259,7 +259,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
|
||||
async def _async_log_event_base(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
verbose_logger.debug(f"s3 Logging - Enters logging function for model {kwargs}")
|
||||
verbose_logger.debug("s3 Logging - Enters logging function for model %s", kwargs)
|
||||
|
||||
s3_batch_logging_element = self.create_s3_batch_logging_element(
|
||||
start_time=start_time,
|
||||
|
|
@ -284,7 +284,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
self.batch_size,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"s3 Layer Error - {e!s}")
|
||||
verbose_logger.exception("s3 Layer Error - %s", e)
|
||||
self.handle_callback_failure(callback_name="S3Logger")
|
||||
|
||||
async def async_upload_data_to_s3(self, batch_logging_element: s3BatchLoggingElement):
|
||||
|
|
@ -313,8 +313,8 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
aws_sts_endpoint=self.s3_aws_sts_endpoint,
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"s3_v2 logger - uploading data to s3 - {batch_logging_element.s3_object_key}")
|
||||
verbose_logger.debug(f"s3_v2 logger - s3_verify setting: {self.s3_verify}")
|
||||
verbose_logger.debug("s3_v2 logger - uploading data to s3 - %s", batch_logging_element.s3_object_key)
|
||||
verbose_logger.debug("s3_v2 logger - s3_verify setting: %s", self.s3_verify)
|
||||
|
||||
# Prepare the URL
|
||||
url = f"https://{self.s3_bucket_name}.s3.{self.s3_region_name}.amazonaws.com/{batch_logging_element.s3_object_key}"
|
||||
|
|
@ -374,16 +374,19 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
if response.status_code in (500, 503) and attempt < max_retries - 1:
|
||||
wait_time = 2**attempt # 1s, 2s
|
||||
verbose_logger.warning(
|
||||
f"S3 upload returned {response.status_code}, retrying in {wait_time}s "
|
||||
f"(attempt {attempt + 1}/{max_retries}) "
|
||||
f"key={batch_logging_element.s3_object_key}"
|
||||
"S3 upload returned %s, retrying in %ss (attempt %s/%s) key=%s",
|
||||
response.status_code,
|
||||
wait_time,
|
||||
attempt + 1,
|
||||
max_retries,
|
||||
batch_logging_element.s3_object_key,
|
||||
)
|
||||
await asyncio.sleep(wait_time)
|
||||
continue
|
||||
response.raise_for_status()
|
||||
break
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error uploading to s3: {e!s}")
|
||||
verbose_logger.exception("Error uploading to s3: %s", e)
|
||||
self.handle_callback_failure(callback_name="S3Logger")
|
||||
|
||||
async def async_send_batch(self):
|
||||
|
|
@ -395,7 +398,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
|
||||
Raises: Does not raise an exception, will only verbose_logger.exception()
|
||||
"""
|
||||
verbose_logger.debug(f"s3_v2 logger - sending batch of {len(self.log_queue)}")
|
||||
verbose_logger.debug("s3_v2 logger - sending batch of %s", len(self.log_queue))
|
||||
if not self.log_queue:
|
||||
return
|
||||
|
||||
|
|
@ -447,7 +450,10 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
|
||||
s3_file_name = litellm.utils.get_logging_id(start_time, standard_logging_payload) or ""
|
||||
verbose_logger.debug(
|
||||
f"Creating s3 file with prefix_components={prefix_components},prefix_path={prefix_path} and {s3_file_name}"
|
||||
"Creating s3 file with prefix_components=%s,prefix_path=%s and %s",
|
||||
prefix_components,
|
||||
prefix_path,
|
||||
s3_file_name,
|
||||
)
|
||||
s3_object_key = get_s3_object_key(
|
||||
s3_path=cast(str | None, self.s3_path) or "",
|
||||
|
|
@ -455,7 +461,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
start_time=start_time,
|
||||
s3_file_name=s3_file_name,
|
||||
)
|
||||
verbose_logger.debug(f"s3_object_key={s3_object_key}")
|
||||
verbose_logger.debug("s3_object_key=%s", s3_object_key)
|
||||
|
||||
s3_object_download_filename = (
|
||||
f"time-{start_time.strftime('%Y-%m-%dT%H-%M-%S-%f')}_{standard_logging_payload['id']}.json"
|
||||
|
|
@ -479,7 +485,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
except ImportError:
|
||||
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
|
||||
try:
|
||||
verbose_logger.debug(f"s3_v2 logger - uploading data to s3 - {batch_logging_element.s3_object_key}")
|
||||
verbose_logger.debug("s3_v2 logger - uploading data to s3 - %s", batch_logging_element.s3_object_key)
|
||||
credentials: Credentials = self.get_credentials(
|
||||
aws_access_key_id=self.s3_aws_access_key_id,
|
||||
aws_secret_access_key=self.s3_aws_secret_access_key,
|
||||
|
|
@ -548,16 +554,19 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
if response.status_code in (500, 503) and attempt < max_retries - 1:
|
||||
wait_time = 2**attempt # 1s, 2s
|
||||
verbose_logger.warning(
|
||||
f"S3 upload returned {response.status_code}, retrying in {wait_time}s "
|
||||
f"(attempt {attempt + 1}/{max_retries}) "
|
||||
f"key={batch_logging_element.s3_object_key}"
|
||||
"S3 upload returned %s, retrying in %ss (attempt %s/%s) key=%s",
|
||||
response.status_code,
|
||||
wait_time,
|
||||
attempt + 1,
|
||||
max_retries,
|
||||
batch_logging_element.s3_object_key,
|
||||
)
|
||||
time.sleep(wait_time)
|
||||
continue
|
||||
response.raise_for_status()
|
||||
break
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error uploading to s3: {e!s}")
|
||||
verbose_logger.exception("Error uploading to s3: %s", e)
|
||||
self.handle_callback_failure(callback_name="S3Logger")
|
||||
|
||||
async def _download_object_from_s3(self, s3_object_key: str) -> dict | None:
|
||||
|
|
@ -596,7 +605,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
aws_sts_endpoint=self.s3_aws_sts_endpoint,
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"s3_v2 logger - downloading data from s3 - {s3_object_key}")
|
||||
verbose_logger.debug("s3_v2 logger - downloading data from s3 - %s", s3_object_key)
|
||||
|
||||
# Prepare the URL
|
||||
url = f"https://{self.s3_bucket_name}.s3.{self.s3_region_name}.amazonaws.com/{s3_object_key}"
|
||||
|
|
@ -642,7 +651,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
return response.json()
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error downloading from S3: {e!s}")
|
||||
verbose_logger.exception("Error downloading from S3: %s", e)
|
||||
return None
|
||||
|
||||
async def get_proxy_server_request_from_cold_storage_with_object_key(
|
||||
|
|
@ -666,5 +675,5 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
|
|||
downloaded_object = await self._download_object_from_s3(object_key)
|
||||
return downloaded_object
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error retrieving object {object_key} from cold storage: {e!s}")
|
||||
verbose_logger.exception("Error retrieving object %s from cold storage: %s", object_key, e)
|
||||
return None
|
||||
|
|
|
|||
|
|
@ -68,7 +68,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
|
|||
**kwargs,
|
||||
) -> None:
|
||||
try:
|
||||
verbose_logger.debug(f"in init sqs logger - sqs_callback_params {litellm.aws_sqs_callback_params}")
|
||||
verbose_logger.debug("in init sqs logger - sqs_callback_params %s", litellm.aws_sqs_callback_params)
|
||||
|
||||
self.async_httpx_client = get_async_httpx_client(
|
||||
llm_provider=httpxSpecialProvider.LoggingCallback,
|
||||
|
|
@ -100,7 +100,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
|
|||
asyncio.create_task(self.periodic_flush())
|
||||
self.flush_lock = asyncio.Lock()
|
||||
|
||||
verbose_logger.debug(f"sqs flush interval: {sqs_flush_interval}, sqs batch size: {sqs_batch_size}")
|
||||
verbose_logger.debug("sqs flush interval: %s, sqs batch size: %s", sqs_flush_interval, sqs_batch_size)
|
||||
|
||||
CustomBatchLogger.__init__(
|
||||
self,
|
||||
|
|
@ -113,7 +113,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
|
|||
BaseAWSLLM.__init__(self)
|
||||
|
||||
except Exception as e:
|
||||
print_verbose(f"Got exception on init sqs client {e!s}")
|
||||
print_verbose(f"Got exception on init sqs client {e}")
|
||||
raise e
|
||||
|
||||
def _init_sqs_params(
|
||||
|
|
@ -215,7 +215,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
|
|||
self.batch_size,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"sqs Layer Error - {e!s}")
|
||||
verbose_logger.exception("sqs Layer Error - %s", e)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
|
|
@ -233,10 +233,10 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
|
|||
)
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Datadog Layer Error - {e!s}\n{traceback.format_exc()}")
|
||||
verbose_logger.exception("Datadog Layer Error - %s\n%s", e, traceback.format_exc())
|
||||
|
||||
async def async_send_batch(self) -> None:
|
||||
verbose_logger.debug(f"sqs logger - sending batch of {len(self.log_queue)}")
|
||||
verbose_logger.debug("sqs logger - sending batch of %s", len(self.log_queue))
|
||||
if not self.log_queue:
|
||||
return
|
||||
|
||||
|
|
@ -305,7 +305,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
|
|||
)
|
||||
response.raise_for_status()
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error sending to SQS: {e!s}")
|
||||
verbose_logger.exception("Error sending to SQS: %s", e)
|
||||
|
||||
async def async_health_check(self) -> IntegrationHealthCheckStatus:
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -15,7 +15,9 @@ class TraceloopLogger:
|
|||
from traceloop.sdk.tracing.tracing import TracerWrapper
|
||||
except ModuleNotFoundError as e:
|
||||
verbose_logger.error(
|
||||
f"Traceloop not installed, try running 'pip install traceloop-sdk' to fix this error: {e}\n{traceback.format_exc()}"
|
||||
"Traceloop not installed, try running 'pip install traceloop-sdk' to fix this error: %s\n%s",
|
||||
e,
|
||||
traceback.format_exc(),
|
||||
)
|
||||
raise e
|
||||
|
||||
|
|
|
|||
|
|
@ -124,7 +124,7 @@ class VectorStorePreCallHook(CustomLogger):
|
|||
},
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"search_response: {search_response}")
|
||||
verbose_logger.debug("search_response: %s", search_response)
|
||||
|
||||
# Store search results for later use in citations
|
||||
all_search_results.append(search_response)
|
||||
|
|
@ -137,7 +137,7 @@ class VectorStorePreCallHook(CustomLogger):
|
|||
# Get the number of results for logging
|
||||
num_results = 0
|
||||
num_results = len(search_response.get("data", []) or [])
|
||||
verbose_logger.debug(f"Vector store search completed. Added context from {num_results} results")
|
||||
verbose_logger.debug("Vector store search completed. Added context from %s results", num_results)
|
||||
|
||||
# Store search results as-is (already in OpenAI-compatible format)
|
||||
if litellm_logging_obj and all_search_results:
|
||||
|
|
@ -146,7 +146,7 @@ class VectorStorePreCallHook(CustomLogger):
|
|||
return model, modified_messages, non_default_params
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error in VectorStorePreCallHook: {e!s}")
|
||||
verbose_logger.exception("Error in VectorStorePreCallHook: %s", e)
|
||||
# Return original parameters on error
|
||||
return model, messages, non_default_params
|
||||
|
||||
|
|
@ -243,14 +243,14 @@ class VectorStorePreCallHook(CustomLogger):
|
|||
verbose_logger.debug("No litellm_logging_obj in request_data")
|
||||
return None
|
||||
|
||||
verbose_logger.debug(f"model_call_details keys: {list(litellm_logging_obj.model_call_details.keys())}")
|
||||
verbose_logger.debug("model_call_details keys: %s", list(litellm_logging_obj.model_call_details.keys()))
|
||||
|
||||
# Get search results from model_call_details (already in OpenAI format)
|
||||
search_results: list[VectorStoreSearchResponse] | None = litellm_logging_obj.model_call_details.get(
|
||||
"search_results"
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"Search results found: {search_results is not None}")
|
||||
verbose_logger.debug("Search results found: %s", search_results is not None)
|
||||
|
||||
if not search_results:
|
||||
verbose_logger.debug("No search results found")
|
||||
|
|
@ -269,13 +269,13 @@ class VectorStorePreCallHook(CustomLogger):
|
|||
# Set the provider_specific_fields
|
||||
setattr(choice.message, "provider_specific_fields", provider_fields)
|
||||
|
||||
verbose_logger.debug(f"Added {len(search_results)} search results to response")
|
||||
verbose_logger.debug("Added %s search results to response", len(search_results))
|
||||
|
||||
# Return modified response
|
||||
return response
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error adding search results to response: {e!s}")
|
||||
verbose_logger.exception("Error adding search results to response: %s", e)
|
||||
# Don't fail the request if search results fail to be added
|
||||
return None
|
||||
|
||||
|
|
@ -297,7 +297,7 @@ class VectorStorePreCallHook(CustomLogger):
|
|||
# Get search results from model_call_details (already in OpenAI format)
|
||||
search_results: list[VectorStoreSearchResponse] | None = request_data.get("search_results")
|
||||
|
||||
verbose_logger.debug(f"Search results found for streaming chunk: {search_results is not None}")
|
||||
verbose_logger.debug("Search results found for streaming chunk: %s", search_results is not None)
|
||||
|
||||
if not search_results:
|
||||
verbose_logger.debug("No search results found for streaming chunk")
|
||||
|
|
@ -316,12 +316,12 @@ class VectorStorePreCallHook(CustomLogger):
|
|||
# Set the provider_specific_fields
|
||||
choice.delta.provider_specific_fields = provider_fields
|
||||
|
||||
verbose_logger.debug(f"Added {len(search_results)} search results to streaming chunk")
|
||||
verbose_logger.debug("Added %s search results to streaming chunk", len(search_results))
|
||||
|
||||
# Return modified chunk
|
||||
return response_chunk
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.exception(f"Error adding search results to streaming chunk: {e!s}")
|
||||
verbose_logger.exception("Error adding search results to streaming chunk: %s", e)
|
||||
# Don't fail the request if search results fail to be added
|
||||
return response_chunk
|
||||
|
|
|
|||
|
|
@ -148,10 +148,10 @@ def get_weave_otel_config() -> WeaveOtelConfig:
|
|||
host = "https://" + host
|
||||
# Self-managed instances use a different path
|
||||
endpoint = host.rstrip("/") + WEAVE_OTEL_ENDPOINT
|
||||
verbose_logger.debug(f"Using Weave OTEL endpoint from host: {endpoint}")
|
||||
verbose_logger.debug("Using Weave OTEL endpoint from host: %s", endpoint)
|
||||
else:
|
||||
endpoint = WEAVE_BASE_URL + WEAVE_OTEL_ENDPOINT
|
||||
verbose_logger.debug(f"Using Weave cloud endpoint: {endpoint}")
|
||||
verbose_logger.debug("Using Weave cloud endpoint: %s", endpoint)
|
||||
|
||||
# Weave uses Basic auth with format: api:<WANDB_API_KEY>
|
||||
auth_header = _get_weave_authorization_header(api_key=api_key)
|
||||
|
|
|
|||
|
|
@ -155,8 +155,8 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
)
|
||||
if anthropic_config is not None and anthropic_config.handles_web_search_natively():
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Skipping short-circuit for {provider_str} "
|
||||
"(provider handles web search natively via the agentic loop)"
|
||||
"WebSearchInterception: Skipping short-circuit for %s (provider handles web search natively via the agentic loop)",
|
||||
provider_str,
|
||||
)
|
||||
return None
|
||||
except (ValueError, Exception):
|
||||
|
|
@ -176,7 +176,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
return None
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Short-circuit search detected (provider={provider_str}, query='{query}')"
|
||||
"WebSearchInterception: Short-circuit search detected (provider=%s, query='%s')", provider_str, query
|
||||
)
|
||||
|
||||
# Native clients (Claude Desktop / Cowork / Anthropic SDK) make a
|
||||
|
|
@ -198,7 +198,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
else:
|
||||
search_result_text, structured = await self._execute_search(query, kwargs=kwargs)
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"WebSearchInterception: Short-circuit search failed: {e}")
|
||||
verbose_logger.error("WebSearchInterception: Short-circuit search failed: %s", e)
|
||||
search_result_text, structured = f"Search failed: {e}", None
|
||||
|
||||
content: list[dict[str, object]] = []
|
||||
|
|
@ -224,7 +224,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
content.append({"type": "text", "text": search_result_text})
|
||||
|
||||
response: dict[str, object] = {
|
||||
"id": f"msg_{uuid.uuid4()!s}",
|
||||
"id": f"msg_{uuid.uuid4()}",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": model,
|
||||
|
|
@ -235,9 +235,9 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
}
|
||||
|
||||
verbose_logger.debug(
|
||||
"WebSearchInterception: Short-circuit search completed, "
|
||||
f"returning synthetic response ({len(search_result_text)} chars, "
|
||||
f"native_blocks={native_tool is not None})"
|
||||
"WebSearchInterception: Short-circuit search completed, returning synthetic response (%s chars, native_blocks=%s)",
|
||||
len(search_result_text),
|
||||
native_tool is not None,
|
||||
)
|
||||
return response
|
||||
|
||||
|
|
@ -294,8 +294,10 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
converted_tool = get_litellm_web_search_tool_openai()
|
||||
converted_tools.append(converted_tool)
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Converted {tool.get('name', 'unknown')} "
|
||||
f"(type={tool.get('type', 'none')}) to {LITELLM_WEB_SEARCH_TOOL_NAME}"
|
||||
"WebSearchInterception: Converted %s (type=%s) to %s",
|
||||
tool.get("name", "unknown"),
|
||||
tool.get("type", "none"),
|
||||
LITELLM_WEB_SEARCH_TOOL_NAME,
|
||||
)
|
||||
else:
|
||||
# Keep other tools as-is
|
||||
|
|
@ -419,14 +421,14 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
custom_llm_provider = kwargs.get("litellm_params", {}).get("custom_llm_provider", "")
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Pre-request hook called"
|
||||
f" - custom_llm_provider={custom_llm_provider}"
|
||||
f" - enabled_providers={self.enabled_providers or 'ALL'}"
|
||||
"WebSearchInterception: Pre-request hook called - custom_llm_provider=%s - enabled_providers=%s",
|
||||
custom_llm_provider,
|
||||
self.enabled_providers or "ALL",
|
||||
)
|
||||
|
||||
if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers:
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Skipping - provider {custom_llm_provider} not in {self.enabled_providers}"
|
||||
"WebSearchInterception: Skipping - provider %s not in %s", custom_llm_provider, self.enabled_providers
|
||||
)
|
||||
return None
|
||||
|
||||
|
|
@ -440,7 +442,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
if not has_websearch:
|
||||
return None
|
||||
|
||||
verbose_logger.debug(f"WebSearchInterception: Pre-request hook triggered for provider={custom_llm_provider}")
|
||||
verbose_logger.debug("WebSearchInterception: Pre-request hook triggered for provider=%s", custom_llm_provider)
|
||||
|
||||
# If the client sent an Anthropic-native web_search_* tool, mark the
|
||||
# request so the agentic loop emits native web_search_tool_result
|
||||
|
|
@ -457,15 +459,17 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
standard_tool = get_litellm_web_search_tool()
|
||||
converted_tools.append(standard_tool)
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Converted {tool.get('name', 'unknown')} "
|
||||
f"(type={tool.get('type', 'none')}) to {LITELLM_WEB_SEARCH_TOOL_NAME}"
|
||||
"WebSearchInterception: Converted %s (type=%s) to %s",
|
||||
tool.get("name", "unknown"),
|
||||
tool.get("type", "none"),
|
||||
LITELLM_WEB_SEARCH_TOOL_NAME,
|
||||
)
|
||||
else:
|
||||
converted_tools.append(tool)
|
||||
|
||||
kwargs["tools"] = converted_tools
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Tools after conversion: {[t.get('name') for t in converted_tools]}"
|
||||
"WebSearchInterception: Tools after conversion: %s", [t.get("name") for t in converted_tools]
|
||||
)
|
||||
|
||||
if "tool_choice" in kwargs:
|
||||
|
|
@ -511,15 +515,17 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
kwargs=kwargs,
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"WebSearchInterception: Hook called! provider={custom_llm_provider}, stream={stream}")
|
||||
verbose_logger.debug(f"WebSearchInterception: Response type: {type(response)}")
|
||||
verbose_logger.debug("WebSearchInterception: Hook called! provider=%s, stream=%s", custom_llm_provider, stream)
|
||||
verbose_logger.debug("WebSearchInterception: Response type: %s", type(response))
|
||||
|
||||
# Check if provider should be intercepted
|
||||
# Note: custom_llm_provider is already normalized by get_llm_provider()
|
||||
# (e.g., "bedrock/invoke/..." -> "bedrock")
|
||||
if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers:
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Skipping provider {custom_llm_provider} (not in enabled list: {self.enabled_providers})"
|
||||
"WebSearchInterception: Skipping provider %s (not in enabled list: %s)",
|
||||
custom_llm_provider,
|
||||
self.enabled_providers,
|
||||
)
|
||||
return False, {}
|
||||
|
||||
|
|
@ -541,7 +547,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
return False, {}
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Detected {len(tool_calls)} WebSearch tool call(s), executing agentic loop"
|
||||
"WebSearchInterception: Detected %s WebSearch tool call(s), executing agentic loop", len(tool_calls)
|
||||
)
|
||||
|
||||
# Extract thinking blocks from response content.
|
||||
|
|
@ -576,7 +582,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
|
||||
if thinking_blocks:
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Extracted {len(thinking_blocks)} thinking block(s) from response"
|
||||
"WebSearchInterception: Extracted %s thinking block(s) from response", len(thinking_blocks)
|
||||
)
|
||||
|
||||
# Return tools dict with tool calls and thinking blocks
|
||||
|
|
@ -606,14 +612,16 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
"""
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Chat completion hook called! provider={custom_llm_provider}, stream={stream}"
|
||||
"WebSearchInterception: Chat completion hook called! provider=%s, stream=%s", custom_llm_provider, stream
|
||||
)
|
||||
verbose_logger.debug(f"WebSearchInterception: Response type: {type(response)}")
|
||||
verbose_logger.debug("WebSearchInterception: Response type: %s", type(response))
|
||||
|
||||
# Check if provider should be intercepted
|
||||
if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers:
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Skipping provider {custom_llm_provider} (not in enabled list: {self.enabled_providers})"
|
||||
"WebSearchInterception: Skipping provider %s (not in enabled list: %s)",
|
||||
custom_llm_provider,
|
||||
self.enabled_providers,
|
||||
)
|
||||
return False, {}
|
||||
|
||||
|
|
@ -635,7 +643,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
return False, {}
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Detected {len(tool_calls)} WebSearch tool call(s), executing agentic loop"
|
||||
"WebSearchInterception: Detected %s WebSearch tool call(s), executing agentic loop", len(tool_calls)
|
||||
)
|
||||
|
||||
# Return tools dict with tool calls
|
||||
|
|
@ -659,12 +667,14 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
) -> tuple[bool, dict]:
|
||||
"""Check if WebSearch interception is needed for the Responses API."""
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Responses hook called! provider={custom_llm_provider}, stream={stream}"
|
||||
"WebSearchInterception: Responses hook called! provider=%s, stream=%s", custom_llm_provider, stream
|
||||
)
|
||||
|
||||
if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers:
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Skipping provider {custom_llm_provider} (not in enabled list: {self.enabled_providers})"
|
||||
"WebSearchInterception: Skipping provider %s (not in enabled list: %s)",
|
||||
custom_llm_provider,
|
||||
self.enabled_providers,
|
||||
)
|
||||
return False, {}
|
||||
|
||||
|
|
@ -684,7 +694,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
return False, {}
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Detected {len(tool_calls)} WebSearch function_call(s), executing agentic loop"
|
||||
"WebSearchInterception: Detected %s WebSearch function_call(s), executing agentic loop", len(tool_calls)
|
||||
)
|
||||
|
||||
tools_dict = {
|
||||
|
|
@ -716,7 +726,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
tool_calls = tools["tool_calls"]
|
||||
thinking_blocks = tools.get("thinking_blocks", [])
|
||||
|
||||
verbose_logger.debug(f"WebSearchInterception: Executing agentic loop for {len(tool_calls)} search(es)")
|
||||
verbose_logger.debug("WebSearchInterception: Executing agentic loop for %s search(es)", len(tool_calls))
|
||||
|
||||
return await self._execute_agentic_loop(
|
||||
model=model,
|
||||
|
|
@ -853,7 +863,8 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
# Object refused write — fall through and leave the response
|
||||
# untouched rather than crash the request.
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: could not inject native blocks into response of type {type(response).__name__}"
|
||||
"WebSearchInterception: could not inject native blocks into response of type %s",
|
||||
type(response).__name__,
|
||||
)
|
||||
return response
|
||||
|
||||
|
|
@ -878,7 +889,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
response_format = tools.get("response_format", "openai")
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Executing chat completion agentic loop for {len(tool_calls)} search(es)"
|
||||
"WebSearchInterception: Executing chat completion agentic loop for %s search(es)", len(tool_calls)
|
||||
)
|
||||
|
||||
return await self._execute_chat_completion_agentic_loop(
|
||||
|
|
@ -962,7 +973,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
for tool_call in tool_calls
|
||||
]
|
||||
|
||||
verbose_logger.debug(f"WebSearchInterception: Executing {len(search_tasks)} responses search(es) in parallel")
|
||||
verbose_logger.debug("WebSearchInterception: Executing %s responses search(es) in parallel", len(search_tasks))
|
||||
search_results = await asyncio.gather(*search_tasks, return_exceptions=True)
|
||||
|
||||
search_texts = [self._extract_search_text(result) for result in search_results]
|
||||
|
|
@ -1038,12 +1049,12 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
@staticmethod
|
||||
def _extract_search_text(result: object) -> str:
|
||||
if isinstance(result, Exception):
|
||||
verbose_logger.error(f"WebSearchInterception: Responses search failed with error: {result!s}")
|
||||
return f"Search failed: {result!s}"
|
||||
verbose_logger.error("WebSearchInterception: Responses search failed with error: %s", result)
|
||||
return f"Search failed: {result}"
|
||||
if isinstance(result, tuple) and len(result) == 2:
|
||||
text_value, _ = result
|
||||
return text_value if isinstance(text_value, str) else str(text_value)
|
||||
verbose_logger.debug(f"WebSearchInterception: Unexpected search result type {type(result)}")
|
||||
verbose_logger.debug("WebSearchInterception: Unexpected search result type %s", type(result))
|
||||
return str(result)
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -1176,15 +1187,15 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
for tool_call in tool_calls:
|
||||
query = tool_call["input"].get("query")
|
||||
if query:
|
||||
verbose_logger.debug(f"WebSearchInterception: Queuing search for query='{query}'")
|
||||
verbose_logger.debug("WebSearchInterception: Queuing search for query='%s'", query)
|
||||
search_tasks.append(self._execute_search(query, kwargs=kwargs))
|
||||
else:
|
||||
verbose_logger.debug(f"WebSearchInterception: Tool call {tool_call['id']} has no query")
|
||||
verbose_logger.debug("WebSearchInterception: Tool call %s has no query", tool_call["id"])
|
||||
# Add empty result for tools without query
|
||||
search_tasks.append(self._create_empty_search_result())
|
||||
|
||||
# Execute searches in parallel
|
||||
verbose_logger.debug(f"WebSearchInterception: Executing {len(search_tasks)} search(es) in parallel")
|
||||
verbose_logger.debug("WebSearchInterception: Executing %s search(es) in parallel", len(search_tasks))
|
||||
search_results = await asyncio.gather(*search_tasks, return_exceptions=True)
|
||||
|
||||
# Split the gathered (text, structured) tuples into two parallel lists.
|
||||
|
|
@ -1194,8 +1205,8 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
structured_results: list[SearchResponse | None] = []
|
||||
for i, result in enumerate(search_results):
|
||||
if isinstance(result, Exception):
|
||||
verbose_logger.error(f"WebSearchInterception: Search {i} failed with error: {result!s}")
|
||||
final_search_results.append(f"Search failed: {result!s}")
|
||||
verbose_logger.error("WebSearchInterception: Search %s failed with error: %s", i, result)
|
||||
final_search_results.append(f"Search failed: {result}")
|
||||
structured_results.append(None)
|
||||
elif isinstance(result, tuple) and len(result) == 2:
|
||||
text_value, structured_value = result
|
||||
|
|
@ -1204,7 +1215,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
else:
|
||||
# Defensive: legacy callers / unexpected shape — preserve text,
|
||||
# drop structure.
|
||||
verbose_logger.debug(f"WebSearchInterception: Unexpected result type {type(result)} at index {i}")
|
||||
verbose_logger.debug("WebSearchInterception: Unexpected result type %s at index %s", type(result), i)
|
||||
final_search_results.append(str(result))
|
||||
structured_results.append(None)
|
||||
|
||||
|
|
@ -1224,7 +1235,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
|
||||
max_tokens = self._resolve_max_tokens(anthropic_messages_optional_request_params, kwargs)
|
||||
|
||||
verbose_logger.debug(f"WebSearchInterception: Using max_tokens={max_tokens} for follow-up request")
|
||||
verbose_logger.debug("WebSearchInterception: Using max_tokens=%s for follow-up request", max_tokens)
|
||||
|
||||
optional_params_without_max_tokens = {
|
||||
k: v for k, v in anthropic_messages_optional_request_params.items() if k != "max_tokens"
|
||||
|
|
@ -1286,12 +1297,12 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
if not search_provider:
|
||||
search_provider = "perplexity"
|
||||
verbose_logger.debug(
|
||||
"WebSearchInterception: No search tools configured in router, "
|
||||
f"using default provider '{search_provider}'"
|
||||
"WebSearchInterception: No search tools configured in router, using default provider '%s'",
|
||||
search_provider,
|
||||
)
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Executing search for '{query}' using provider '{search_provider}'"
|
||||
"WebSearchInterception: Executing search for '%s' using provider '%s'", query, search_provider
|
||||
)
|
||||
search_kwargs = {
|
||||
key: value
|
||||
|
|
@ -1304,11 +1315,11 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
search_result_text = WebSearchTransformation.format_search_response(result)
|
||||
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Search completed for '{query}', got {len(search_result_text)} chars"
|
||||
"WebSearchInterception: Search completed for '%s', got %s chars", query, len(search_result_text)
|
||||
)
|
||||
return search_result_text, result
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"WebSearchInterception: Search failed for '{query}': {e!s}")
|
||||
verbose_logger.error("WebSearchInterception: Search failed for '%s': %s", query, e)
|
||||
raise
|
||||
|
||||
async def _authorize_search_tool(
|
||||
|
|
@ -1392,21 +1403,25 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
if matching_tools:
|
||||
search_provider = (matching_tools[0].get("litellm_params", {}) or {}).get("search_provider")
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Found search tool '{self.search_tool_name}' "
|
||||
f"from {source} with provider '{search_provider}'"
|
||||
"WebSearchInterception: Found search tool '%s' from %s with provider '%s'",
|
||||
self.search_tool_name,
|
||||
source,
|
||||
search_provider,
|
||||
)
|
||||
return matching_tools[0]
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Search tool '{self.search_tool_name}' not found in {source}, "
|
||||
"falling back to first available or perplexity"
|
||||
"WebSearchInterception: Search tool '%s' not found in %s, falling back to first available or perplexity",
|
||||
self.search_tool_name,
|
||||
source,
|
||||
)
|
||||
|
||||
if search_tools:
|
||||
first_tool = search_tools[0]
|
||||
search_provider = (first_tool.get("litellm_params", {}) or {}).get("search_provider")
|
||||
verbose_logger.debug(
|
||||
f"WebSearchInterception: Using first available search tool from {source} "
|
||||
f"with provider '{search_provider}'"
|
||||
"WebSearchInterception: Using first available search tool from %s with provider '%s'",
|
||||
source,
|
||||
search_provider,
|
||||
)
|
||||
return first_tool
|
||||
|
||||
|
|
@ -1470,15 +1485,15 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
query = args.get("query")
|
||||
|
||||
if query:
|
||||
verbose_logger.debug(f"WebSearchInterception: Queuing search for query='{query}'")
|
||||
verbose_logger.debug("WebSearchInterception: Queuing search for query='%s'", query)
|
||||
search_tasks.append(self._execute_search(query, kwargs=kwargs))
|
||||
else:
|
||||
verbose_logger.debug(f"WebSearchInterception: Tool call {tool_call.get('id')} has no query")
|
||||
verbose_logger.debug("WebSearchInterception: Tool call %s has no query", tool_call.get("id"))
|
||||
# Add empty result for tools without query
|
||||
search_tasks.append(self._create_empty_search_result())
|
||||
|
||||
# Execute searches in parallel
|
||||
verbose_logger.debug(f"WebSearchInterception: Executing {len(search_tasks)} search(es) in parallel")
|
||||
verbose_logger.debug("WebSearchInterception: Executing %s search(es) in parallel", len(search_tasks))
|
||||
search_results = await asyncio.gather(*search_tasks, return_exceptions=True)
|
||||
|
||||
# Chat-completion path only needs text — OpenAI tool_result format
|
||||
|
|
@ -1486,13 +1501,13 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
final_search_results: list[str] = []
|
||||
for i, result in enumerate(search_results):
|
||||
if isinstance(result, Exception):
|
||||
verbose_logger.error(f"WebSearchInterception: Search {i} failed with error: {result!s}")
|
||||
final_search_results.append(f"Search failed: {result!s}")
|
||||
verbose_logger.error("WebSearchInterception: Search %s failed with error: %s", i, result)
|
||||
final_search_results.append(f"Search failed: {result}")
|
||||
elif isinstance(result, tuple) and len(result) == 2:
|
||||
text_value, _ = result
|
||||
final_search_results.append(cast(str, text_value) if isinstance(text_value, str) else str(text_value))
|
||||
else:
|
||||
verbose_logger.debug(f"WebSearchInterception: Unexpected result type {type(result)} at index {i}")
|
||||
verbose_logger.debug("WebSearchInterception: Unexpected result type %s at index %s", type(result), i)
|
||||
final_search_results.append(str(result))
|
||||
|
||||
# Build assistant and tool messages using transformation
|
||||
|
|
@ -1517,7 +1532,7 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
]
|
||||
|
||||
verbose_logger.debug("WebSearchInterception: Making follow-up chat completion request with search results")
|
||||
verbose_logger.debug(f"WebSearchInterception: Follow-up messages count: {len(follow_up_messages)}")
|
||||
verbose_logger.debug("WebSearchInterception: Follow-up messages count: %s", len(follow_up_messages))
|
||||
|
||||
# Remove internal parameters that shouldn't be passed to follow-up request
|
||||
internal_params = {
|
||||
|
|
|
|||
|
|
@ -103,7 +103,7 @@ class WebSearchTransformation:
|
|||
parsed_input = json.loads(arguments) if arguments else {}
|
||||
except json.JSONDecodeError:
|
||||
verbose_logger.warning(
|
||||
f"WebSearchInterception: Failed to parse function_call arguments: {arguments}"
|
||||
"WebSearchInterception: Failed to parse function_call arguments: %s", arguments
|
||||
)
|
||||
parsed_input = {}
|
||||
elif isinstance(arguments, dict):
|
||||
|
|
@ -122,7 +122,7 @@ class WebSearchTransformation:
|
|||
"input": parsed_input,
|
||||
}
|
||||
)
|
||||
verbose_logger.debug(f"WebSearchInterception: Found {item_name} function_call with call_id={call_id}")
|
||||
verbose_logger.debug("WebSearchInterception: Found %s function_call with call_id=%s", item_name, call_id)
|
||||
|
||||
return len(tool_calls) > 0, tool_calls
|
||||
|
||||
|
|
@ -178,7 +178,7 @@ class WebSearchTransformation:
|
|||
"input": block_input,
|
||||
}
|
||||
tool_calls.append(tool_call)
|
||||
verbose_logger.debug(f"WebSearchInterception: Found {block_name} tool_use with id={tool_call['id']}")
|
||||
verbose_logger.debug("WebSearchInterception: Found %s tool_use with id=%s", block_name, tool_call["id"])
|
||||
|
||||
return len(tool_calls) > 0, tool_calls
|
||||
|
||||
|
|
@ -255,7 +255,7 @@ class WebSearchTransformation:
|
|||
arguments = json.loads(function_arguments)
|
||||
except json.JSONDecodeError:
|
||||
verbose_logger.warning(
|
||||
f"WebSearchInterception: Failed to parse function arguments: {function_arguments}"
|
||||
"WebSearchInterception: Failed to parse function arguments: %s", function_arguments
|
||||
)
|
||||
arguments = {}
|
||||
else:
|
||||
|
|
@ -273,7 +273,7 @@ class WebSearchTransformation:
|
|||
"input": arguments, # For compatibility with Anthropic format
|
||||
}
|
||||
tool_calls.append(tool_call_dict)
|
||||
verbose_logger.debug(f"WebSearchInterception: Found {function_name} tool_call with id={tool_id}")
|
||||
verbose_logger.debug("WebSearchInterception: Found %s tool_call with id=%s", function_name, tool_id)
|
||||
|
||||
return len(tool_calls) > 0, tool_calls
|
||||
|
||||
|
|
|
|||
|
|
@ -42,9 +42,9 @@ try:
|
|||
elif response["object"] == "chat.completion":
|
||||
return self._resolve_chat_completion(request, response, time_elapsed)
|
||||
else:
|
||||
logger.debug(f"Unknown OpenAI response object: {response['object']}")
|
||||
logger.debug("Unknown OpenAI response object: %s", response["object"])
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to resolve request/response: {e}")
|
||||
logger.warning("Failed to resolve request/response: %s", e)
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue