Merge origin/litellm_internal_staging into litellm_lit4395_cursor_agent

This commit is contained in:
mateo-berri 2026-08-04 10:20:03 -07:00
commit 9eeff06263
1122 changed files with 21336 additions and 6074 deletions

View file

@ -88,6 +88,36 @@ commands:
rm -f /tmp/uv-install.sh
echo 'export PATH="$HOME/.local/bin:$PATH"' >> "$BASH_ENV"
export PATH="$HOME/.local/bin:$PATH"
install_rust:
description: "Install pinned rustup (1.28.2) and Rust toolchain (1.97.1) with checksum verification. Adds ~/.cargo/bin to PATH. Run this before any `uv sync` or `uv build` of the workspace: the root package builds litellm-rust through maturin, and on an image without cargo maturin fetches an unpinned rustup and a floating toolchain by itself."
steps:
- run:
name: Install Rust (rustup 1.28.2, toolchain 1.97.1)
command: |
case "$(uname -m)" in
x86_64)
RUSTUP_TRIPLE=x86_64-unknown-linux-gnu
RUSTUP_SHA256=20a06e644b0d9bd2fbdbfd52d42540bdde820ea7df86e92e533c073da0cdd43c
;;
aarch64)
RUSTUP_TRIPLE=aarch64-unknown-linux-gnu
RUSTUP_SHA256=e3853c5a252fca15252d07cb23a1bdd9377a8c6f3efa01531109281ae47f841c
;;
*)
echo "install_rust: unsupported architecture $(uname -m)" >&2
exit 1
;;
esac
curl -sSLf -o /tmp/rustup-init \
"https://static.rust-lang.org/rustup/archive/1.28.2/${RUSTUP_TRIPLE}/rustup-init"
echo "${RUSTUP_SHA256} /tmp/rustup-init" | sha256sum -c -
chmod +x /tmp/rustup-init
/tmp/rustup-init -y --no-modify-path --profile minimal --default-toolchain 1.97.1
rm -f /tmp/rustup-init
echo 'export PATH="$HOME/.cargo/bin:$PATH"' >> "$BASH_ENV"
export PATH="$HOME/.cargo/bin:$PATH"
rustc --version
cargo --version
start_postgres:
description: "Start a postgres-db container on port 5432 and wait until it accepts connections."
parameters:
@ -163,6 +193,26 @@ commands:
done
echo "fake OpenAI endpoint did not become ready" >&2
exit 1
start_cost_center_service:
description: "Start the stand-in cost center validation service (tests/store_model_in_db_tests/cost_center_service.py) on host port 9414 and wait until healthy. The proxy's team-metadata validator (team_metadata_validator_e2e.py, impl 'http') reaches it via TEAM_METADATA_VALIDATION_SERVICE_URL=http://host.docker.internal:9414/validate. Run after uv deps are synced."
steps:
- run:
name: Start cost center validation service
background: true
command: |
uv run --no-sync python tests/store_model_in_db_tests/cost_center_service.py --host 0.0.0.0 --port 9414
- run:
name: Wait for cost center validation service
command: |
for i in $(seq 1 30); do
if curl -sf http://localhost:9414/health >/dev/null 2>&1; then
echo "cost center validation service is up"
exit 0
fi
sleep 1
done
echo "cost center validation service did not become ready" >&2
exit 1
setup_litellm_enterprise_pip:
steps:
- run:
@ -178,6 +228,7 @@ commands:
- checkout
- setup_google_dns
- install_uv
- install_rust
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
@ -292,6 +343,7 @@ jobs:
- checkout
- setup_google_dns
- install_uv
- install_rust
- run:
name: Build the wheel
environment:
@ -324,6 +376,7 @@ jobs:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -397,6 +450,7 @@ jobs:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -471,6 +525,7 @@ jobs:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -522,6 +577,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -588,6 +644,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -628,6 +685,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -669,6 +727,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -702,6 +761,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
@ -752,6 +812,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
@ -803,6 +864,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -836,6 +898,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
@ -882,6 +945,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -928,6 +992,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -970,6 +1035,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1016,6 +1082,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1063,6 +1130,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
@ -1103,6 +1171,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1148,6 +1217,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1192,6 +1262,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1224,6 +1295,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1267,6 +1339,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1311,6 +1384,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1355,6 +1429,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1386,6 +1461,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1432,6 +1508,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1477,6 +1554,7 @@ jobs:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1527,6 +1605,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1551,6 +1630,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1577,6 +1657,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1678,6 +1759,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1773,6 +1855,7 @@ jobs:
at: ~/project
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1861,6 +1944,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -1944,6 +2028,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -2076,6 +2161,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -2162,6 +2248,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -2258,12 +2345,14 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
uv sync --frozen --all-groups --all-extras --python 3.12
- start_postgres
- start_fake_openai_endpoint
- start_cost_center_service
- attach_workspace:
at: ~/project
- run:
@ -2283,11 +2372,13 @@ jobs:
-e STORE_MODEL_IN_DB="True" \
-e LITELLM_MASTER_KEY="sk-1234" \
-e FAKE_OPENAI_API_BASE=http://host.docker.internal:8190 \
-e TEAM_METADATA_VALIDATION_SERVICE_URL=http://host.docker.internal:9414/validate \
-e LITELLM_LICENSE=$LITELLM_LICENSE \
-e LITELLM_LOG=ERROR \
--add-host host.docker.internal:host-gateway \
--name my-app \
-v $(pwd)/litellm/proxy/example_config_yaml/store_model_db_config.yaml:/app/config.yaml \
-v $(pwd)/litellm/proxy/example_config_yaml/team_metadata_validator_e2e.py:/app/team_metadata_validator_e2e.py \
litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000
@ -2333,6 +2424,7 @@ jobs:
- setup_google_dns
# Remove Docker CLI installation since it's already available in machine executor
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -2414,6 +2506,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -2553,6 +2646,7 @@ jobs:
- skip_if_unrelated_changes
- setup_google_dns
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
@ -2743,6 +2837,7 @@ jobs:
category: client
- setup_google_dns
- install_uv
- install_rust
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
@ -2885,6 +2980,7 @@ jobs:
category: client
- setup_google_dns
- install_uv
- install_rust
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}

View file

@ -239,7 +239,7 @@ lint-dev: lint-format-changed check-circular-imports check-import-safety
# test-linting.yml (Python), test-litellm-ui-build.yml's frontend-lint (dashboard), and
# check-ui-api-types.yml (API-type drift), skipping any whose files you didn't stage.
# Not auto-installed as a git hook so it never slows an unrelated human commit.
pre-commit:
pre-commit: bootstrap
./scripts/pre_commit_lint.sh
# Testing targets

View file

@ -0,0 +1,17 @@
-- AlterTable
ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
-- AlterTable
ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
-- AlterTable
ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
-- AlterTable
ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
-- AlterTable
ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;
-- AlterTable
ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0;

View file

@ -748,6 +748,7 @@ model LiteLLM_DailyUserSpend {
compression_saved_tokens BigInt @default(0)
compression_savings_spend Float @default(0.0)
prompt_caching_savings_spend Float @default(0.0)
autorouter_savings_spend Float @default(0.0)
spend Float @default(0.0)
api_requests BigInt @default(0)
successful_requests BigInt @default(0)
@ -782,6 +783,7 @@ model LiteLLM_DailyOrganizationSpend {
compression_saved_tokens BigInt @default(0)
compression_savings_spend Float @default(0.0)
prompt_caching_savings_spend Float @default(0.0)
autorouter_savings_spend Float @default(0.0)
spend Float @default(0.0)
api_requests BigInt @default(0)
successful_requests BigInt @default(0)
@ -816,6 +818,7 @@ model LiteLLM_DailyEndUserSpend {
compression_saved_tokens BigInt @default(0)
compression_savings_spend Float @default(0.0)
prompt_caching_savings_spend Float @default(0.0)
autorouter_savings_spend Float @default(0.0)
spend Float @default(0.0)
api_requests BigInt @default(0)
successful_requests BigInt @default(0)
@ -849,6 +852,7 @@ model LiteLLM_DailyAgentSpend {
compression_saved_tokens BigInt @default(0)
compression_savings_spend Float @default(0.0)
prompt_caching_savings_spend Float @default(0.0)
autorouter_savings_spend Float @default(0.0)
spend Float @default(0.0)
api_requests BigInt @default(0)
successful_requests BigInt @default(0)
@ -882,6 +886,7 @@ model LiteLLM_DailyTeamSpend {
compression_saved_tokens BigInt @default(0)
compression_savings_spend Float @default(0.0)
prompt_caching_savings_spend Float @default(0.0)
autorouter_savings_spend Float @default(0.0)
spend Float @default(0.0)
api_requests BigInt @default(0)
successful_requests BigInt @default(0)
@ -917,6 +922,7 @@ model LiteLLM_DailyTagSpend {
compression_saved_tokens BigInt @default(0)
compression_savings_spend Float @default(0.0)
prompt_caching_savings_spend Float @default(0.0)
autorouter_savings_spend Float @default(0.0)
spend Float @default(0.0)
api_requests BigInt @default(0)
successful_requests BigInt @default(0)

View file

@ -264,6 +264,7 @@ databricks_key: Optional[str] = None
openai_like_key: Optional[str] = None
azure_key: Optional[str] = None
anthropic_key: Optional[str] = None
autorouter_savings_baseline_model: Optional[str] = None
replicate_key: Optional[str] = None
bytez_key: Optional[str] = None
gdc_key: Optional[str] = None

View file

@ -23,6 +23,7 @@ from typing import Any, cast
# Import all the data structures that define what can be lazy-loaded
# These are just lists of names and maps of where to find them
from ._lazy_imports_registry import (
# Import maps
_BEDROCK_TYPES_IMPORT_MAP,
_CACHING_IMPORT_MAP,
_COST_CALCULATOR_IMPORT_MAP,
@ -33,12 +34,11 @@ from ._lazy_imports_registry import (
_TOKEN_COUNTER_IMPORT_MAP,
_TYPES_IMPORT_MAP,
_TYPES_UTILS_IMPORT_MAP,
# Import maps
_UTILS_IMPORT_MAP,
_UTILS_MODULE_IMPORT_MAP,
# Name tuples
BEDROCK_TYPES_NAMES,
CACHING_NAMES,
# Name tuples
COST_CALCULATOR_NAMES,
DOTPROMPT_NAMES,
HTTP_HANDLER_NAMES,

View file

@ -651,7 +651,9 @@ def get_redis_async_client(
if arg in args:
url_kwargs[arg] = redis_kwargs[arg]
else:
verbose_logger.debug(f"REDIS: ignoring argument: {arg}. Not an allowed async_redis.Redis.from_url arg.")
verbose_logger.debug(
"REDIS: ignoring argument: %s. Not an allowed async_redis.Redis.from_url arg.", arg
)
return async_redis.Redis.from_url(**url_kwargs)
# Check for Redis Sentinel
@ -805,6 +807,6 @@ def _pretty_print_redis_config(redis_kwargs: dict) -> None:
# Fallback to simple logging if rich is not available
masker = SensitiveDataMasker()
masked_redis_kwargs = masker.mask_dict(redis_kwargs)
verbose_logger.info(f"Redis configuration: {masked_redis_kwargs}")
verbose_logger.info("Redis configuration: %s", masked_redis_kwargs)
except Exception as e:
verbose_logger.error(f"Error pretty printing Redis configuration: {e}")
verbose_logger.error("Error pretty printing Redis configuration: %s", e)

View file

@ -148,13 +148,13 @@ class LiteLLMA2ACardResolver(_A2ACardResolver): # type: ignore[misc]
last_error = None
for path in paths:
try:
verbose_logger.debug(f"Attempting to fetch agent card from {self.base_url}{path}")
verbose_logger.debug("Attempting to fetch agent card from %s%s", self.base_url, path)
return await super().get_agent_card(
relative_card_path=path,
http_kwargs=http_kwargs,
)
except Exception as e:
verbose_logger.debug(f"Failed to fetch agent card from {self.base_url}{path}: {e}")
verbose_logger.debug("Failed to fetch agent card from %s%s: %s", self.base_url, path, e)
last_error = e
continue

View file

@ -192,9 +192,11 @@ async def handle_a2a_localhost_retry(
request_type = "streaming " if is_streaming else ""
verbose_logger.warning(
f"A2A {request_type}request to '{error.localhost_url}' failed: {error.original_error}. "
f"Agent card contains localhost/internal URL. "
f"Retrying with base_url '{error.base_url}'."
"A2A %srequest to '%s' failed: %s. Agent card contains localhost/internal URL. Retrying with base_url '%s'.",
request_type,
error.localhost_url,
error.original_error,
error.base_url,
)
# Fix the agent card URL

View file

@ -76,7 +76,7 @@ class A2ACompletionBridgeHandler:
)
if a2a_provider_config is not None:
verbose_logger.info(f"A2A: Using provider config for {custom_llm_provider}")
verbose_logger.info("A2A: Using provider config for %s", custom_llm_provider)
return await a2a_provider_config.handle_non_streaming(
request_id=request_id,
@ -103,7 +103,7 @@ class A2ACompletionBridgeHandler:
else:
full_model = model
verbose_logger.info(f"A2A completion bridge: model={full_model}, api_base={api_base}")
verbose_logger.info("A2A completion bridge: model=%s, api_base=%s", full_model, api_base)
# Build completion params dict
completion_params: dict[str, Any] = {
@ -143,7 +143,7 @@ class A2ACompletionBridgeHandler:
request_id=request_id,
)
verbose_logger.info(f"A2A completion bridge completed: request_id={request_id}")
verbose_logger.info("A2A completion bridge completed: request_id=%s", request_id)
return a2a_response
@ -185,7 +185,7 @@ class A2ACompletionBridgeHandler:
)
if a2a_provider_config is not None:
verbose_logger.info(f"A2A: Using provider config for {custom_llm_provider} (streaming)")
verbose_logger.info("A2A: Using provider config for %s (streaming)", custom_llm_provider)
async for chunk in a2a_provider_config.handle_streaming(
request_id=request_id,
@ -221,7 +221,7 @@ class A2ACompletionBridgeHandler:
else:
full_model = model
verbose_logger.info(f"A2A completion bridge streaming: model={full_model}, api_base={api_base}")
verbose_logger.info("A2A completion bridge streaming: model=%s, api_base=%s", full_model, api_base)
# Build completion params dict
completion_params: dict[str, Any] = {
@ -300,7 +300,9 @@ class A2ACompletionBridgeHandler:
)
yield completed_event
verbose_logger.info(f"A2A completion bridge streaming completed: request_id={request_id}, chunks={chunk_count}")
verbose_logger.info(
"A2A completion bridge streaming completed: request_id=%s, chunks=%s", request_id, chunk_count
)
# Convenience functions that delegate to the class methods

View file

@ -109,7 +109,7 @@ class A2ACompletionBridgeTransformation:
extra_body = {**extra_body, "metadata": merged_metadata}
completion_params["extra_body"] = extra_body
verbose_logger.debug(f"A2A -> completion forward metadata keys={list(forward_metadata.keys())}")
verbose_logger.debug("A2A -> completion forward metadata keys=%s", list(forward_metadata.keys()))
@staticmethod
def a2a_message_to_openai_messages(
@ -145,7 +145,9 @@ class A2ACompletionBridgeTransformation:
# once at run level via extra_body.metadata (LangGraph POST /runs/wait shape).
openai_message: dict[str, Any] = {"role": openai_role, "content": content}
verbose_logger.debug(f"A2A -> OpenAI transform: role={role} -> {openai_role}, content_length={len(content)}")
verbose_logger.debug(
"A2A -> OpenAI transform: role=%s -> %s, content_length=%s", role, openai_role, len(content)
)
return [openai_message]
@ -186,7 +188,7 @@ class A2ACompletionBridgeTransformation:
"result": a2a_message,
}
verbose_logger.debug(f"OpenAI -> A2A transform: content_length={len(content)}")
verbose_logger.debug("OpenAI -> A2A transform: content_length=%s", len(content))
return a2a_response

View file

@ -204,7 +204,7 @@ async def _send_message_via_completion_bridge(
Requires request; api_base is optional for providers that derive endpoint from model.
"""
verbose_logger.info(f"A2A using completion bridge: provider={custom_llm_provider}, api_base={api_base}")
verbose_logger.info("A2A using completion bridge: provider=%s, api_base=%s", custom_llm_provider, api_base)
from litellm.a2a_protocol.litellm_completion_bridge.handler import (
A2ACompletionBridgeHandler,
@ -463,7 +463,7 @@ async def asend_message(
agent_name = _get_a2a_model_info(a2a_client, kwargs)
verbose_logger.info(f"A2A send_message request_id={request.id}, agent={agent_name}")
verbose_logger.info("A2A send_message request_id=%s, agent=%s", request.id, agent_name)
# Get agent card URL for localhost retry logic
agent_card = _get_a2a_client_agent_card(a2a_client)
@ -478,7 +478,7 @@ async def asend_message(
agent_name=agent_name,
)
verbose_logger.info(f"A2A send_message completed, request_id={request.id}")
verbose_logger.info("A2A send_message completed, request_id=%s", request.id)
# Wrap in LiteLLM response type for _hidden_params support
response = LiteLLMSendMessageResponse.from_a2a_response(a2a_response, request_id=str(request.id))
@ -640,7 +640,7 @@ async def asend_message_streaming(
raise ValueError("request is required for completion bridge")
# api_base is optional for providers that derive endpoint from model (e.g., bedrock/agentcore)
verbose_logger.info(f"A2A streaming using completion bridge: provider={custom_llm_provider}")
verbose_logger.info("A2A streaming using completion bridge: provider=%s", custom_llm_provider)
from litellm.a2a_protocol.litellm_completion_bridge.handler import (
A2ACompletionBridgeHandler,
@ -697,7 +697,7 @@ async def asend_message_streaming(
proxy_server_request=proxy_server_request,
)
verbose_logger.info(f"A2A send_message_streaming request_id={request.id}, agent={agent_name}")
verbose_logger.info("A2A send_message_streaming request_id=%s, agent=%s", request.id, agent_name)
agent_card = _get_a2a_client_agent_card(a2a_client)
card_url = get_agent_card_url(agent_card) if agent_card else None
@ -759,7 +759,7 @@ async def create_a2a_client(
"The 'a2a' package is required for A2A agent invocation. Install it with: pip install a2a-sdk"
)
verbose_logger.info(f"Creating A2A client for {base_url}")
verbose_logger.info("Creating A2A client for %s", base_url)
# Use get_async_httpx_client with per-agent params so that different agents
# (with different extra_headers) get separate cached clients. The params
@ -781,7 +781,7 @@ async def create_a2a_client(
httpx_client = _async_handler.client
if extra_headers:
httpx_client.headers.update(extra_headers)
verbose_proxy_logger.debug(f"A2A client created with extra_headers={list(extra_headers.keys())}")
verbose_proxy_logger.debug("A2A client created with extra_headers=%s", list(extra_headers.keys()))
a2a_client = await create_client( # pyright: ignore[reportOptionalCall]
base_url,
@ -798,7 +798,7 @@ async def create_a2a_client(
if agent_card is not None:
a2a_client._litellm_agent_card = agent_card # type: ignore[attr-defined]
verbose_logger.info(f"A2A client created for {base_url}")
verbose_logger.info("A2A client created for %s", base_url)
return a2a_client
@ -824,7 +824,7 @@ async def aget_agent_card(
"The 'a2a' package is required for A2A agent invocation. Install it with: pip install a2a-sdk"
)
verbose_logger.info(f"Fetching agent card from {base_url}")
verbose_logger.info("Fetching agent card from %s", base_url)
# Use LiteLLM's cached httpx client
http_handler = get_async_httpx_client(
@ -839,5 +839,5 @@ async def aget_agent_card(
)
agent_card = await resolver.get_agent_card()
verbose_logger.info(f"Fetched agent card: {agent_card.name if hasattr(agent_card, 'name') else 'unknown'}")
verbose_logger.info("Fetched agent card: %s", agent_card.name if hasattr(agent_card, "name") else "unknown")
return agent_card

View file

@ -53,7 +53,7 @@ class BedrockAgentCoreA2AHandler:
agent_extra_headers=agent_extra_headers,
)
verbose_logger.info(f"BedrockAgentCore A2A: Sending non-streaming request to {url}")
verbose_logger.info("BedrockAgentCore A2A: Sending non-streaming request to %s", url)
client = get_async_httpx_client(
llm_provider=cast(Any, httpxSpecialProvider.A2AProvider),
@ -67,7 +67,7 @@ class BedrockAgentCoreA2AHandler:
response_data = response.json()
if "error" in response_data:
verbose_logger.warning(f"BedrockAgentCore A2A: Agent returned error: {response_data['error']}")
verbose_logger.warning("BedrockAgentCore A2A: Agent returned error: %s", response_data["error"])
return response_data
@ -100,7 +100,7 @@ class BedrockAgentCoreA2AHandler:
agent_extra_headers=agent_extra_headers,
)
verbose_logger.info(f"BedrockAgentCore A2A: Sending streaming request to {url}")
verbose_logger.info("BedrockAgentCore A2A: Sending streaming request to %s", url)
client = get_async_httpx_client(
llm_provider=cast(Any, httpxSpecialProvider.A2AProvider),

View file

@ -195,5 +195,5 @@ class BedrockAgentCoreA2ATransformation:
event = json.loads(data_str)
yield event
except json.JSONDecodeError:
verbose_logger.debug(f"BedrockAgentCore A2A: Skipping non-JSON SSE line: {data_str[:100]}")
verbose_logger.debug("BedrockAgentCore A2A: Skipping non-JSON SSE line: %s", data_str[:100])
continue

View file

@ -47,7 +47,7 @@ class PydanticAIHandler:
"""
if api_base is None:
raise ValueError("api_base is required for Pydantic AI agents")
verbose_logger.info(f"Pydantic AI: Routing to Pydantic AI agent at {api_base}")
verbose_logger.info("Pydantic AI: Routing to Pydantic AI agent at %s", api_base)
# Send request directly to Pydantic AI agent
response_data = await PydanticAITransformation.send_non_streaming_request(
@ -92,7 +92,7 @@ class PydanticAIHandler:
"""
if api_base is None:
raise ValueError("api_base is required for Pydantic AI agents")
verbose_logger.info(f"Pydantic AI: Faking streaming for Pydantic AI agent at {api_base}")
verbose_logger.info("Pydantic AI: Faking streaming for Pydantic AI agent at %s", api_base)
# Get raw task response first (not the transformed A2A format)
raw_response = await PydanticAITransformation.send_and_get_raw_response(

View file

@ -118,7 +118,7 @@ class PydanticAITransformation:
status = result.get("status", {})
state = status.get("state", "")
verbose_logger.debug(f"Pydantic AI: Poll attempt {attempt + 1}/{max_attempts}, state={state}")
verbose_logger.debug("Pydantic AI: Poll attempt %s/%s, state=%s", attempt + 1, max_attempts, state)
if state == "completed":
return poll_data
@ -173,7 +173,7 @@ class PydanticAITransformation:
# FastA2A uses root endpoint (/) not /messages
endpoint = api_base.rstrip("/")
verbose_logger.info(f"Pydantic AI: Sending non-streaming request to {endpoint}")
verbose_logger.info("Pydantic AI: Sending non-streaming request to %s", endpoint)
# Send request to Pydantic AI agent using shared async HTTP client
client = get_async_httpx_client(
@ -200,7 +200,7 @@ class PydanticAITransformation:
# Need to poll for completion
task_id = result.get("id")
if task_id:
verbose_logger.info(f"Pydantic AI: Task {task_id} submitted, polling for completion...")
verbose_logger.info("Pydantic AI: Task %s submitted, polling for completion...", task_id)
response_data = await PydanticAITransformation._poll_for_completion(
client=client,
endpoint=endpoint,
@ -209,7 +209,7 @@ class PydanticAITransformation:
agent_extra_headers=agent_extra_headers,
)
verbose_logger.info(f"Pydantic AI: Received completed response for request_id={request_id}")
verbose_logger.info("Pydantic AI: Received completed response for request_id=%s", request_id)
return response_data
@ -518,4 +518,4 @@ class PydanticAITransformation:
}
yield completed_event
verbose_logger.info(f"Pydantic AI: Fake streaming completed for request_id={request_id}")
verbose_logger.info("Pydantic AI: Fake streaming completed for request_id=%s", request_id)

View file

@ -135,7 +135,7 @@ class WatsonxOrchestrateHandler:
response.raise_for_status()
result: dict[str, Any] = response.json()
status = result.get("status", "")
verbose_logger.debug(f"WXO: Poll {attempt + 1}/{max_attempts} run='{run_id}' status='{status}'")
verbose_logger.debug("WXO: Poll %s/%s run='%s' status='%s'", attempt + 1, max_attempts, run_id, status)
if status in WatsonxOrchestrateTransformation.TERMINAL_STATES:
return result
@ -297,8 +297,8 @@ class WatsonxOrchestrateHandler:
response.raise_for_status()
except httpx.TransportError as exc:
verbose_logger.warning(
f"WXO: Streaming request failed before a run was submitted "
f"({exc!r}), falling back to non-streaming + fake streaming",
"WXO: Streaming request failed before a run was submitted (%r), falling back to non-streaming + fake streaming",
exc,
exc_info=True,
)
result = await WatsonxOrchestrateHandler.handle_non_streaming(

View file

@ -214,4 +214,4 @@ class WatsonxOrchestrateTransformation:
},
}
verbose_logger.debug(f"WXO: Fake streaming completed for request_id={request_id}")
verbose_logger.debug("WXO: Fake streaming completed for request_id=%s", request_id)

View file

@ -138,13 +138,15 @@ class A2AStreamingIterator:
)
verbose_logger.info(
f"A2A streaming completed: prompt_tokens={prompt_tokens}, "
f"completion_tokens={completion_tokens}, total_tokens={total_tokens}, "
f"response_cost={response_cost}"
"A2A streaming completed: prompt_tokens=%s, completion_tokens=%s, total_tokens=%s, response_cost=%s",
prompt_tokens,
completion_tokens,
total_tokens,
response_cost,
)
except Exception as e:
verbose_logger.debug(f"Error in A2A streaming completion handler: {e}")
verbose_logger.debug("Error in A2A streaming completion handler: %s", e)
def _build_logging_result(self, usage: litellm.Usage) -> dict[str, Any]:
"""Build a result dict for logging."""

View file

@ -51,7 +51,7 @@ class GetAnthropicBetaHeadersConfig:
)
return content
except Exception as e:
verbose_logger.error(f"Failed to load local beta headers config: {e}")
verbose_logger.error("Failed to load local beta headers config: %s", e)
# Return empty config as fallback
return {
"anthropic": {},
@ -246,7 +246,9 @@ def filter_and_transform_beta_headers(
# Check if header is in the mapping
if header not in provider_mapping:
verbose_logger.debug(f"Dropping unknown beta header '{header}' for provider '{provider}' (not in mapping)")
verbose_logger.debug(
"Dropping unknown beta header '%s' for provider '%s' (not in mapping)", header, provider
)
continue
# Get the mapped header value
@ -254,7 +256,7 @@ def filter_and_transform_beta_headers(
# Skip if header is unsupported (null value)
if mapped_header is None:
verbose_logger.debug(f"Dropping unsupported beta header '{header}' for provider '{provider}'")
verbose_logger.debug("Dropping unsupported beta header '%s' for provider '%s'", header, provider)
continue
# Add the mapped header

View file

@ -249,7 +249,7 @@ def batch_completion_models_all_responses(*args, **kwargs):
if result is not None:
responses.append(result)
except Exception as e:
print_verbose(f"batch_completion_models_all_responses: model request failed: {e!s}")
print_verbose(f"batch_completion_models_all_responses: model request failed: {e}")
continue
return responses

View file

@ -258,10 +258,10 @@ async def _fetch_batch_output_file_content(
if is_base64_unified_file_id:
try:
file_id = is_base64_unified_file_id.split("llm_output_file_id,")[1].split(";")[0]
verbose_logger.debug(f"Extracted LLM output file ID from unified file ID: {file_id}")
verbose_logger.debug("Extracted LLM output file ID from unified file ID: %s", file_id)
except (IndexError, AttributeError) as e:
verbose_logger.error(
f"Failed to extract LLM output file ID from unified file ID: {batch.output_file_id}, error: {e}"
"Failed to extract LLM output file ID from unified file ID: %s, error: %s", batch.output_file_id, e
)
# Build kwargs for afile_content with credentials from litellm_params

View file

@ -182,7 +182,7 @@ def create_batch(
)
except Exception as e:
verbose_logger.exception(
f"litellm.batches.main.py::create_batch() - Error inferring custom_llm_provider - {e!s}"
"litellm.batches.main.py::create_batch() - Error inferring custom_llm_provider - %s", e
)
_is_async = kwargs.pop("acreate_batch", False) is True
@ -890,7 +890,7 @@ def cancel_batch(
)
except Exception as e:
verbose_logger.exception(
f"litellm.batches.main.py::cancel_batch() - Error inferring custom_llm_provider - {e!s}"
"litellm.batches.main.py::cancel_batch() - Error inferring custom_llm_provider - %s", e
)
optional_params = GenericLiteLLMParams(**kwargs)
litellm_params = get_litellm_params(

View file

@ -67,7 +67,10 @@ class AzureBlobCache(BaseCache):
cached_response = json.loads(as_str)
verbose_logger.debug(
f"Got Azure Blob Cache: key: {key}, cached_response {cached_response}. Type Response {type(cached_response)}"
"Got Azure Blob Cache: key: %s, cached_response %s. Type Response %s",
key,
cached_response,
type(cached_response),
)
return cached_response
@ -84,7 +87,10 @@ class AzureBlobCache(BaseCache):
as_str = as_bytes.decode("utf-8")
cached_response = json.loads(as_str)
verbose_logger.debug(
f"Got Azure Blob Cache: key: {key}, cached_response {cached_response}. Type Response {type(cached_response)}"
"Got Azure Blob Cache: key: %s, cached_response %s. Type Response %s",
key,
cached_response,
type(cached_response),
)
return cached_response
except ResourceNotFoundError:

View file

@ -353,13 +353,13 @@ class Cache:
if param in combined_kwargs:
param_value: str | None = self._get_param_value(param, kwargs)
if param_value is not None:
cache_key += f"{param!s}: {param_value!s}"
cache_key += f"{param}: {param_value}"
elif param not in litellm_param_kwargs: # check if user passed in optional param - e.g. top_k
if litellm.enable_caching_on_provider_specific_optional_params is True: # feature flagged for now
if kwargs[param] is None:
continue # ignore None params
param_value = kwargs[param]
cache_key += f"{param!s}: {param_value!s}"
cache_key += f"{param}: {param_value}"
if is_semantic_cache:
cache_key += self._get_semantic_cache_tenant_scope(kwargs)
@ -676,7 +676,7 @@ class Cache:
cache_key, cached_data, kwargs = self._add_cache_logic(result=result, **kwargs)
self.cache.set_cache(cache_key, cached_data, **kwargs)
except Exception as e:
verbose_logger.exception(f"LiteLLM Cache: Excepton add_cache: {e!s}")
verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e)
async def async_add_cache(self, result, dynamic_cache_object: BaseCache | None = None, **kwargs):
"""
@ -695,7 +695,7 @@ class Cache:
else:
await self.cache.async_set_cache(cache_key, cached_data, **kwargs)
except Exception as e:
verbose_logger.exception(f"LiteLLM Cache: Excepton add_cache: {e!s}")
verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e)
def _convert_to_cached_embedding(
self,
@ -874,7 +874,7 @@ class Cache:
else:
await self.cache.async_set_cache_pipeline(cache_list=cache_list, **kwargs)
except Exception as e:
verbose_logger.exception(f"LiteLLM Cache: Excepton add_cache: {e!s}")
verbose_logger.exception("LiteLLM Cache: Excepton add_cache: %s", e)
def should_use_cache(self, **kwargs):
"""

View file

@ -271,7 +271,7 @@ class LLMCachingHandler:
embedding_all_elements_cache_hit=embedding_all_elements_cache_hit,
)
verbose_logger.debug(f"CACHE RESULT: {cached_result}")
verbose_logger.debug("CACHE RESULT: %s", cached_result)
return CachingHandlerResponse(
cached_result=cached_result,
final_embedding_cached_response=final_embedding_cached_response,

View file

@ -147,7 +147,7 @@ class DualCache(BaseCache):
return result
except Exception as e:
verbose_logger.error(f"LiteLLM Cache: Excepton async add_cache: {e!s}")
verbose_logger.error("LiteLLM Cache: Excepton async add_cache: %s", e)
raise e
def get_cache(
@ -347,7 +347,7 @@ class DualCache(BaseCache):
if self.redis_cache is not None and local_only is False:
await self.redis_cache.async_set_cache(key, value, **kwargs)
except Exception as e:
verbose_logger.exception(f"LiteLLM Cache: Excepton async add_cache: {e!s}")
verbose_logger.exception("LiteLLM Cache: Excepton async add_cache: %s", e)
# async_batch_set_cache
async def async_set_cache_pipeline(self, cache_list: list, local_only: bool = False, **kwargs):
@ -366,7 +366,7 @@ class DualCache(BaseCache):
cache_list=cache_list, ttl=kwargs.pop("ttl", None), **kwargs
)
except Exception as e:
verbose_logger.exception(f"LiteLLM Cache: Excepton async add_cache: {e!s}")
verbose_logger.exception("LiteLLM Cache: Excepton async add_cache: %s", e)
async def async_increment_cache(
self,

View file

@ -0,0 +1,276 @@
"""
Deferred close of HTTP/SDK clients that the LLM client cache has evicted.
Eviction only drops the cache's reference to a client. Every OpenAI/Azure SDK
client is a reference cycle (each resource namespace holds the client back), so
an evicted client and its pooled TCP connections survive until a generational
collection runs, which under load is thousands of requests later.
Closing at eviction time is not an option: a request that was handed the client
just before it was evicted is still using it, and closing it underneath that
request raises ``RuntimeError: Cannot send a request, as the client has been
closed.``
So an evicted client is closed once two conditions hold. A grace window must
have passed since its eviction, which covers a request that holds the client
but is momentarily not on the wire, and the client must report no connection in
flight. The second condition is what keeps the first honest: a request may run
for ``litellm.request_timeout`` seconds, 6000 by default, and a streaming
response is bounded only by how long the upstream keeps sending, so no deadline
on its own can promise that a request has finished.
Only clients litellm itself created are closed; a client the caller supplied is
left alone because litellm does not own its lifecycle.
A client that closes synchronously is closed from wherever the cache is next
used. One whose close is a coroutine needs the event loop it was evicted on, so
it waits for a call from that loop rather than having work scheduled onto a loop
it does not belong to. Queued clients are therefore bucketed by what it takes to
close them, and each bucket is ordered by deadline, so a reap walks the entries
that are due rather than the whole queue.
The queue holds its clients weakly, so waiting out a grace window never keeps
alive anything the collector would have reclaimed first.
"""
import asyncio
import inspect
import threading
import time
import weakref
from collections import deque
from collections.abc import Awaitable, Callable, Iterator
from dataclasses import dataclass, replace
from litellm.constants import (
EVICTED_LLM_CLIENT_CLOSE_GRACE_SECONDS,
EVICTED_LLM_CLIENT_CLOSE_MAX_PENDING,
)
_CLOSABLE_ANYWHERE = "closable-anywhere"
_CLOSABLE_ON_ANY_LOOP = "closable-on-any-loop"
_BucketKey = str | int
@dataclass(frozen=True, slots=True)
class _PendingClose:
"""A queued close.
The client is held weakly, so queueing one never keeps alive anything the
collector would otherwise have reclaimed first.
``needs_loop`` is set for a client whose close is a coroutine; those can only
be closed from the event loop they were evicted on, recorded in ``loop_id``.
A client that closes synchronously carries neither constraint.
"""
client_ref: "weakref.ref[object]"
loop_id: int | None
needs_loop: bool
close_after: float
def _bucket_key(pending: _PendingClose) -> _BucketKey:
"""Which reaps can close this entry: any at all, any running a loop, or one loop's."""
if not pending.needs_loop:
return _CLOSABLE_ANYWHERE
if pending.loop_id is None:
return _CLOSABLE_ON_ANY_LOOP
return pending.loop_id
def _running_loop_id() -> int | None:
try:
return id(asyncio.get_running_loop())
except RuntimeError:
return None
def _close_function(client: object) -> Callable[[], object] | None:
close_fn: Callable[[], object] | None = getattr(client, "aclose", None) or getattr(client, "close", None)
return close_fn
def _transport_of(client: object) -> object:
"""The httpx transport behind an SDK wrapper, a litellm handler, or a bare client."""
for holder in (getattr(client, "_client", None), getattr(client, "client", None), client):
transport: object = getattr(holder, "_transport", None)
if transport is not None:
return transport
return None
def _connection_is_idle(connection: object) -> bool:
"""A pooled connection is idle unless it is servicing a request."""
is_idle: object = getattr(connection, "is_idle", None)
return bool(is_idle()) if callable(is_idle) else True
def _pool_has_busy_connection(transport: object) -> bool | None:
"""Whether the httpcore pool behind the transport is servicing a request.
``None`` when there is no such pool, so the caller can ask the other backend.
"""
pooled: object = getattr(getattr(transport, "_pool", None), "connections", None)
if not isinstance(pooled, (list, tuple)):
return None
return any(
not _connection_is_idle(connection) # pyright: ignore[reportUnknownArgumentType] # untyped pool list
for connection in pooled # pyright: ignore[reportUnknownVariableType] # untyped pool list
)
def _has_connection_in_flight(client: object) -> bool:
"""Whether the client is servicing a request right now.
Both connection backends litellm uses already account for the connections
they have handed out, so this reads the client's own lease accounting rather
than inferring it from elapsed time: httpcore reports a non-idle connection
for the whole of a response including a stream, and aiohttp holds the
connection in ``_acquired`` over the same span.
A client that cannot answer is reported as idle, which leaves the grace
window as the only guard, exactly as it was before this check existed.
"""
try:
transport = _transport_of(client)
pooled_busy = _pool_has_busy_connection(transport)
if pooled_busy is not None:
return pooled_busy
session: object = getattr(transport, "client", None)
return bool(getattr(getattr(session, "connector", None), "_acquired", None))
except Exception: # noqa: BLE001 - a client that cannot report its state is treated as idle
return False
async def _close_quietly(closing: Awaitable[object]) -> None:
try:
await closing
except Exception: # noqa: BLE001 - a discarded client's close must never surface to callers
pass
class EvictedClientCloser:
"""Closes evicted, litellm-owned clients once they are idle and out of grace."""
def __init__(
self,
grace_seconds: float = EVICTED_LLM_CLIENT_CLOSE_GRACE_SECONDS,
max_pending: int = EVICTED_LLM_CLIENT_CLOSE_MAX_PENDING,
clock: Callable[[], float] = time.monotonic,
) -> None:
self._grace_seconds = grace_seconds
self._max_pending = max_pending
self._clock = clock
self._owned: weakref.WeakSet[object] = weakref.WeakSet()
self._buckets: dict[_BucketKey, deque[_PendingClose]] = {} # mutable-ok: deadline-ordered queues
self._pending_count = 0
self._queue_lock = threading.Lock() # the cache is reachable from every worker thread's loop
self._close_tasks: set[asyncio.Task[None]] = set() # mutable-ok: strong refs to running closes
def mark_owned(self, client: object) -> None:
"""Record that litellm created this client, so it may be closed on eviction."""
try:
self._owned.add(client)
except TypeError:
pass # values that cannot be weak-referenced are never litellm clients
def _is_owned(self, client: object) -> bool:
try:
return client in self._owned
except TypeError:
return False # unhashable values are never litellm clients
def schedule(self, client: object) -> None:
"""Queue an evicted client for closing once it is idle and out of grace.
Past ``max_pending`` the client is left to the collector instead, so a
workload that churns the cache cannot grow this queue without bound.
Every queued entry comes due within one grace window, so the capacity it
occupies is returned within that window rather than held.
"""
if client is None or not self._is_owned(client):
return
close_fn = _close_function(client)
if close_fn is None:
return
if self._pending_count >= self._max_pending:
return
self._enqueue(
_PendingClose(
client_ref=weakref.ref(client),
loop_id=_running_loop_id(),
needs_loop=inspect.iscoroutinefunction(close_fn),
close_after=self._clock() + self._grace_seconds,
)
)
def reap(self) -> None:
"""Close every queued client that is due, idle, and closable from here.
Called from the cache's read path, so the empty-queue exit comes first and
the work done past it is proportional to what is due, not to the queue.
"""
if not self._pending_count:
return
now = self._clock()
for pending in self._take_due(_running_loop_id(), now):
client = pending.client_ref()
if client is None:
continue
if _has_connection_in_flight(client):
self._enqueue(replace(pending, close_after=now + self._grace_seconds))
continue
self._close(client)
@property
def pending_count(self) -> int:
return self._pending_count
def _enqueue(self, pending: _PendingClose) -> None:
"""Append to the entry's bucket, dropping any dead entries it queues behind.
Deadlines only ever move forward, so appending keeps each bucket ordered
by deadline, and entries whose client the collector already took sit at
the front rather than having to be searched for.
"""
with self._queue_lock:
bucket = self._buckets.setdefault(_bucket_key(pending), deque()) # mutable-ok: FIFO by design
while bucket and bucket[0].client_ref() is None:
bucket.popleft()
self._pending_count -= 1
bucket.append(pending)
self._pending_count += 1
def _take_due(self, loop_id: int | None, now: float) -> tuple[_PendingClose, ...]:
buckets = (_CLOSABLE_ANYWHERE,) if loop_id is None else (_CLOSABLE_ANYWHERE, _CLOSABLE_ON_ANY_LOOP, loop_id)
with self._queue_lock:
return tuple(pending for key in buckets for pending in self._drain_locked(key, now))
def _drain_locked(self, key: _BucketKey, now: float) -> Iterator[_PendingClose]:
bucket = self._buckets.get(key)
if bucket is None:
return
while bucket and bucket[0].close_after <= now:
self._pending_count -= 1
yield bucket.popleft()
if not bucket:
del self._buckets[key]
def _close(self, client: object) -> None:
close_fn = _close_function(client)
if close_fn is None:
return
try:
closing = close_fn()
except Exception: # noqa: BLE001 - a discarded client's close must never surface to callers
return
if not inspect.isawaitable(closing):
return
task = asyncio.get_running_loop().create_task(_close_quietly(closing))
self._close_tasks.add(task)
task.add_done_callback(self._close_tasks.discard)
default_evicted_client_closer = EvictedClientCloser()

View file

@ -71,12 +71,15 @@ class GCSCache(BaseCache):
if response.status_code == 200:
cached_response = json.loads(response.text)
verbose_logger.debug(
f"Got GCS Cache: key: {key}, cached_response {cached_response}. Type Response {type(cached_response)}"
"Got GCS Cache: key: %s, cached_response %s. Type Response %s",
key,
cached_response,
type(cached_response),
)
return cached_response
return None
except Exception as e:
verbose_logger.error(f"GCS Caching: get_cache() - Got exception from GCS: {e}")
verbose_logger.error("GCS Caching: get_cache() - Got exception from GCS: %s", e)
async def async_get_cache(self, key, **kwargs):
try:
@ -89,7 +92,7 @@ class GCSCache(BaseCache):
return json.loads(response.text)
return None
except Exception as e:
verbose_logger.error(f"GCS Caching: async_get_cache() - Got exception from GCS: {e}")
verbose_logger.error("GCS Caching: async_get_cache() - Got exception from GCS: %s", e)
def flush_cache(self):
pass

View file

@ -4,21 +4,44 @@ Add the event loop to the cache key, to prevent event loop closed errors.
import asyncio
from .evicted_client_closer import EvictedClientCloser, default_evicted_client_closer
from .in_memory_cache import InMemoryCache
class LLMClientCache(InMemoryCache):
"""Cache for LLM HTTP clients (OpenAI, Azure, httpx, etc.).
IMPORTANT: This cache intentionally does NOT close clients on eviction.
Evicted clients may still be in use by in-flight requests. Closing them
eagerly causes ``RuntimeError: Cannot send a request, as the client has
been closed.`` errors in production after the TTL (1 hour) expires.
An evicted client is never closed on the spot: a request handed the client
just before eviction is still using it, and closing it there raises
``RuntimeError: Cannot send a request, as the client has been closed.``
Clients that are no longer referenced will be garbage-collected normally.
For explicit shutdown cleanup, use ``close_litellm_async_clients()``.
Nor can eviction be left to rely on garbage collection. The SDK clients are
reference cycles, so an evicted client and its open TCP connections survive
until a generational collection runs. Instead a client litellm created is
handed to ``EvictedClientCloser``, which closes it once a grace window has
passed. Clients the caller supplied are left untouched.
"""
def __init__(
self,
max_size_in_memory: int | None = 200,
default_ttl: int | None = 600,
max_size_per_item: int | None = 1024,
evicted_client_closer: EvictedClientCloser | None = None,
):
super().__init__(
max_size_in_memory=max_size_in_memory,
default_ttl=default_ttl,
max_size_per_item=max_size_per_item,
)
self.evicted_client_closer = evicted_client_closer or default_evicted_client_closer
def _remove_key(self, key: str) -> None:
evicted: object = self.cache_dict.get(key)
super()._remove_key(key)
self.evicted_client_closer.schedule(evicted)
self.evicted_client_closer.reap()
def update_cache_key_with_event_loop(self, key):
"""
Add the event loop to the cache key, to prevent event loop closed errors.
@ -31,16 +54,22 @@ class LLMClientCache(InMemoryCache):
except RuntimeError: # handle no current running event loop
return key
def set_cache(self, key, value, **kwargs):
def set_cache(self, key: str, value: object, litellm_owned_client: bool = False, **kwargs):
"""``litellm_owned_client`` marks a client litellm built, so it may be closed once evicted."""
if litellm_owned_client:
self.evicted_client_closer.mark_owned(value)
key = self.update_cache_key_with_event_loop(key)
return super().set_cache(key, value, **kwargs)
async def async_set_cache(self, key, value, **kwargs):
async def async_set_cache(self, key: str, value: object, litellm_owned_client: bool = False, **kwargs):
if litellm_owned_client:
self.evicted_client_closer.mark_owned(value)
key = self.update_cache_key_with_event_loop(key)
return await super().async_set_cache(key, value, **kwargs)
def get_cache(self, key, **kwargs):
key = self.update_cache_key_with_event_loop(key)
self.evicted_client_closer.reap()
return super().get_cache(key, **kwargs)

View file

@ -178,7 +178,7 @@ class QdrantSemanticCache(BaseCache):
if response.status_code not in (200, 201):
print_verbose(f"Qdrant semantic-cache could not create cache-key payload index: {response.text}")
except Exception as exc:
print_verbose(f"Qdrant semantic-cache could not create cache-key payload index: {exc!s}")
print_verbose(f"Qdrant semantic-cache could not create cache-key payload index: {exc}")
def _payload_matches_cache_key(self, payload: dict, key: str) -> bool:
# Pre-isolation points stored only prompt + response with no cache-key

View file

@ -346,7 +346,8 @@ class RedisCache(BaseCache):
verbose_logger.debug("Ignoring async redis ping. No running event loop.")
else:
verbose_logger.error(
f"Error connecting to Async Redis client - {e!s}",
"Error connecting to Async Redis client - %s",
e,
extra={"error": str(e)},
)
self._handle_async_ping_error(e)
@ -483,7 +484,7 @@ class RedisCache(BaseCache):
)
except Exception as e:
# NON blocking - notify users Redis is throwing an exception
print_verbose(f"litellm.caching.caching: set() - Got exception from REDIS : {e!s}")
print_verbose(f"litellm.caching.caching: set() - Got exception from REDIS : {e}")
def increment_cache(self, key, value: int, ttl: float | None = None, **kwargs) -> int:
_redis_client = self.redis_client
@ -1139,7 +1140,7 @@ class RedisCache(BaseCache):
return decoded_results
except Exception as e:
verbose_logger.error(f"Error occurred in batch get cache - {e!s}")
verbose_logger.error("Error occurred in batch get cache - %s", e)
return key_value_dict
@_redis_circuit_breaker_guard
@ -1185,7 +1186,7 @@ class RedisCache(BaseCache):
event_metadata={"key": key},
)
)
print_verbose(f"litellm.caching.caching: async get() - Got exception from REDIS: {e!s}")
print_verbose(f"litellm.caching.caching: async get() - Got exception from REDIS: {e}")
_record_swallowed_redis_failure(self._circuit_breaker, e)
@_redis_circuit_breaker_guard
@ -1257,7 +1258,7 @@ class RedisCache(BaseCache):
parent_otel_span=parent_otel_span,
)
)
verbose_logger.error(f"Error occurred in async batch get cache - {e!s}")
verbose_logger.error("Error occurred in async batch get cache - %s", e)
_record_swallowed_redis_failure(self._circuit_breaker, e)
return key_value_dict
@ -1292,7 +1293,7 @@ class RedisCache(BaseCache):
error=e,
call_type=f"sync_ping <- {_get_call_stack_info()}",
)
verbose_logger.error(f"LiteLLM Redis Cache PING: - Got exception from REDIS : {e!s}")
verbose_logger.error("LiteLLM Redis Cache PING: - Got exception from REDIS : %s", e)
raise e
async def ping(self) -> bool:
@ -1326,7 +1327,7 @@ class RedisCache(BaseCache):
call_type=f"async_ping <- {_get_call_stack_info()}",
)
)
verbose_logger.error(f"LiteLLM Redis Cache PING: - Got exception from REDIS : {e!s}")
verbose_logger.error("LiteLLM Redis Cache PING: - Got exception from REDIS : %s", e)
raise e
@_redis_circuit_breaker_guard
@ -1388,10 +1389,10 @@ class RedisCache(BaseCache):
else:
return {"status": "failed", "message": "Redis ping returned False"}
except Exception as e:
verbose_logger.error(f"Redis connection test failed: {e!s}")
verbose_logger.error("Redis connection test failed: %s", e)
return {
"status": "failed",
"message": f"Redis connection failed: {e!s}",
"message": f"Redis connection failed: {e}",
"error": str(e),
}
@ -1426,7 +1427,7 @@ class RedisCache(BaseCache):
# Execute the pipeline and return results
results = await pipe.execute()
# only return float values
verbose_logger.debug(f"Increment ASYNC Redis Cache PIPELINE: results: {results}")
verbose_logger.debug("Increment ASYNC Redis Cache PIPELINE: results: %s", results)
return [r for r in results if isinstance(r, float)]
@_redis_circuit_breaker_guard
@ -1513,7 +1514,7 @@ class RedisCache(BaseCache):
return None
return ttl
except Exception as e:
verbose_logger.debug(f"Redis TTL Error: {e}")
verbose_logger.debug("Redis TTL Error: %s", e)
_record_swallowed_redis_failure(self._circuit_breaker, e)
return None
@ -1565,7 +1566,7 @@ class RedisCache(BaseCache):
call_type=f"async_rpush <- {_get_call_stack_info()}",
)
)
verbose_logger.error(f"LiteLLM Redis Cache RPUSH: - Got exception from REDIS : {e!s}")
verbose_logger.error("LiteLLM Redis Cache RPUSH: - Got exception from REDIS : %s", e)
raise e
async def _pipeline_rpush_helper(
@ -1711,7 +1712,7 @@ class RedisCache(BaseCache):
call_type=f"async_lpop <- {_get_call_stack_info()}",
)
)
verbose_logger.error(f"LiteLLM Redis Cache LPOP: - Got exception from REDIS : {e!s}")
verbose_logger.error("LiteLLM Redis Cache LPOP: - Got exception from REDIS : %s", e)
raise e
async def _pipeline_lpop_helper(

View file

@ -100,9 +100,9 @@ class RedisClusterCache(RedisCache):
except Exception as e:
from litellm._logging import verbose_logger
verbose_logger.error(f"Redis Cluster connection test failed: {e!s}")
verbose_logger.error("Redis Cluster connection test failed: %s", e)
return {
"status": "failed",
"message": f"Redis Cluster connection failed: {e!s}",
"message": f"Redis Cluster connection failed: {e}",
"error": str(e),
}

View file

@ -138,7 +138,7 @@ class RedisSemanticCache(BaseCache):
cache_vectorizer=cache_vectorizer,
)
except Exception as e:
verbose_logger.error(f"Redis semantic-cache index build failed: {e}")
verbose_logger.error("Redis semantic-cache index build failed: %s", e)
raise
@classmethod
@ -364,7 +364,7 @@ class RedisSemanticCache(BaseCache):
try:
cached_response = ast.literal_eval(cached_response)
except (ValueError, SyntaxError) as e:
print_verbose(f"Error parsing cached response: {e!s}")
print_verbose(f"Error parsing cached response: {e}")
return None
return cached_response
@ -403,7 +403,7 @@ class RedisSemanticCache(BaseCache):
store_kwargs["ttl"] = int(ttl)
self.llmcache.store(prompt, value_str, **store_kwargs)
except Exception as e:
print_verbose(f"Error setting {value_str or value} in the Redis semantic cache: {e!s}")
print_verbose(f"Error setting {value_str or value} in the Redis semantic cache: {e}")
def get_cache(self, key: str, **kwargs) -> Any:
"""
@ -468,7 +468,7 @@ class RedisSemanticCache(BaseCache):
return self._get_cache_logic(cached_response=cached_response)
except Exception as e:
print_verbose(f"Error retrieving from Redis semantic cache: {e!s}")
print_verbose(f"Error retrieving from Redis semantic cache: {e}")
kwargs.setdefault("metadata", {})["semantic-similarity"] = 0.0
async def _get_async_embedding(self, prompt: str, metadata: dict[str, Any] | None = None) -> list[float]:
@ -505,8 +505,8 @@ class RedisSemanticCache(BaseCache):
)
return embedding_response["data"][0]["embedding"]
except Exception as e:
print_verbose(f"Error generating async embedding: {e!s}")
raise ValueError(f"Failed to generate embedding: {e!s}") from e
print_verbose(f"Error generating async embedding: {e}")
raise ValueError(f"Failed to generate embedding: {e}") from e
async def async_set_cache(self, key: str, value: Any, **kwargs) -> None:
"""
@ -546,7 +546,7 @@ class RedisSemanticCache(BaseCache):
**store_kwargs,
)
except Exception as e:
print_verbose(f"Error in async_set_cache: {e!s}")
print_verbose(f"Error in async_set_cache: {e}")
async def async_get_cache(self, key: str, **kwargs) -> Any:
"""
@ -612,7 +612,7 @@ class RedisSemanticCache(BaseCache):
return self._get_cache_logic(cached_response=cached_response)
except Exception as e:
print_verbose(f"Error in async_get_cache: {e!s}")
print_verbose(f"Error in async_get_cache: {e}")
kwargs.setdefault("metadata", {})["semantic-similarity"] = 0.0
async def _index_info(self) -> dict[str, Any]:
@ -639,4 +639,4 @@ class RedisSemanticCache(BaseCache):
tasks.append(self.async_set_cache(val[0], val[1], **kwargs))
await asyncio.gather(*tasks)
except Exception as e:
print_verbose(f"Error in async_set_cache_pipeline: {e!s}")
print_verbose(f"Error in async_set_cache_pipeline: {e}")

View file

@ -104,12 +104,12 @@ class S3Cache(BaseCache):
Compatible with Python 3.8+.
"""
try:
verbose_logger.debug(f"Set ASYNC S3 Cache: Key={key}. Value={value}")
verbose_logger.debug("Set ASYNC S3 Cache: Key=%s. Value=%s", key, value)
loop = asyncio.get_event_loop()
func = partial(self.set_cache, key, value, **kwargs)
await loop.run_in_executor(None, func)
except Exception as e:
verbose_logger.error(f"S3 Caching: async_set_cache() - Got exception from S3: {e}")
verbose_logger.error("S3 Caching: async_set_cache() - Got exception from S3: %s", e)
def get_cache(self, key, **kwargs):
import botocore
@ -138,17 +138,20 @@ class S3Cache(BaseCache):
if not isinstance(cached_response, dict):
cached_response = dict(cached_response)
verbose_logger.debug(
f"Got S3 Cache: key: {key}, cached_response {cached_response}. Type Response {type(cached_response)}"
"Got S3 Cache: key: %s, cached_response %s. Type Response %s",
key,
cached_response,
type(cached_response),
)
return cached_response
except botocore.exceptions.ClientError as e: # type: ignore
if e.response["Error"]["Code"] == "NoSuchKey":
verbose_logger.debug(f"S3 Cache: The specified key '{key}' does not exist in the S3 bucket.")
verbose_logger.debug("S3 Cache: The specified key '%s' does not exist in the S3 bucket.", key)
return None
except Exception as e:
verbose_logger.error(f"S3 Caching: get_cache() - Got exception from S3: {e}")
verbose_logger.error("S3 Caching: get_cache() - Got exception from S3: %s", e)
async def async_get_cache(self, key, **kwargs):
"""
@ -156,13 +159,13 @@ class S3Cache(BaseCache):
Compatible with Python 3.8+.
"""
try:
verbose_logger.debug(f"Get ASYNC S3 Cache: key: {key}")
verbose_logger.debug("Get ASYNC S3 Cache: key: %s", key)
loop = asyncio.get_event_loop()
func = partial(self.get_cache, key, **kwargs)
result = await loop.run_in_executor(None, func)
return result
except Exception as e:
verbose_logger.error(f"S3 Caching: async_get_cache() - Got exception from S3: {e}")
verbose_logger.error("S3 Caching: async_get_cache() - Got exception from S3: %s", e)
return None
def flush_cache(self):

View file

@ -249,7 +249,7 @@ class ValkeySemanticCache(RedisSemanticCache):
if ttl is not None:
self.sync_client.expire(doc_key, ttl)
except Exception as e:
print_verbose(f"Error in Valkey semantic-cache set_cache: {e!s}")
print_verbose(f"Error in Valkey semantic-cache set_cache: {e}")
def get_cache(self, key: str, **kwargs: Any) -> Any:
print_verbose(f"Valkey semantic-cache get_cache, kwargs: {kwargs}")
@ -268,7 +268,7 @@ class ValkeySemanticCache(RedisSemanticCache):
)
return self._resolve_hit(self._first_hit(search_result), key, **kwargs)
except Exception as e:
print_verbose(f"Error in Valkey semantic-cache get_cache: {e!s}")
print_verbose(f"Error in Valkey semantic-cache get_cache: {e}")
kwargs.setdefault("metadata", {})["semantic-similarity"] = 0.0
async def async_set_cache(self, key: str, value: Any, **kwargs: Any) -> None:
@ -288,7 +288,7 @@ class ValkeySemanticCache(RedisSemanticCache):
if ttl is not None:
await self.async_client.expire(doc_key, ttl)
except Exception as e:
print_verbose(f"Error in async Valkey semantic-cache set_cache: {e!s}")
print_verbose(f"Error in async Valkey semantic-cache set_cache: {e}")
async def async_get_cache(self, key: str, **kwargs: Any) -> Any:
print_verbose(f"Async Valkey semantic-cache get_cache, kwargs: {kwargs}")
@ -307,14 +307,14 @@ class ValkeySemanticCache(RedisSemanticCache):
)
return self._resolve_hit(self._first_hit(search_result), key, **kwargs)
except Exception as e:
print_verbose(f"Error in async Valkey semantic-cache get_cache: {e!s}")
print_verbose(f"Error in async Valkey semantic-cache get_cache: {e}")
kwargs.setdefault("metadata", {})["semantic-similarity"] = 0.0
async def async_set_cache_pipeline(self, cache_list: list[tuple[str, Any]], **kwargs: Any) -> None:
try:
await asyncio.gather(*[self.async_set_cache(key, value, **kwargs) for key, value in cache_list])
except Exception as e:
print_verbose(f"Error in Valkey semantic-cache async_set_cache_pipeline: {e!s}")
print_verbose(f"Error in Valkey semantic-cache async_set_cache_pipeline: {e}")
async def _index_info(self) -> dict:
return await self.async_client.ft(self.index_name).info()

View file

@ -465,7 +465,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
self._map_optional_params_to_responses_api_request(optional_params, responses_api_request)
stream = optional_params.get("stream") or litellm_params.get("stream", False)
verbose_logger.debug(f"Chat provider: Stream parameter: {stream}")
verbose_logger.debug("Chat provider: Stream parameter: %s", stream)
# Ensure stream is properly set in the request
if stream:
@ -475,7 +475,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
previous_response_id = optional_params.get("previous_response_id")
if previous_response_id:
# Use the existing session handler for responses API
verbose_logger.debug(f"Chat provider: Warning ignoring previous response ID: {previous_response_id}")
verbose_logger.debug("Chat provider: Warning ignoring previous response ID: %s", previous_response_id)
# Convert back to responses API format for the actual request
@ -495,7 +495,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
"client": client,
}
verbose_logger.debug(f"Chat provider: Final request model={api_model}, input_items={len(input_items)}")
verbose_logger.debug("Chat provider: Final request model=%s, input_items=%s", api_model, len(input_items))
self._merge_responses_api_request_into_request_data(request_data, responses_api_request, instructions)
@ -843,29 +843,29 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
"""Convert chat completion content to responses API format"""
from litellm.types.llms.openai import ChatCompletionImageObject
verbose_logger.debug(f"Chat provider: Converting content to responses format - input type: {type(content)}")
verbose_logger.debug("Chat provider: Converting content to responses format - input type: %s", type(content))
if content is None:
return [self._convert_content_str_to_input_text("", role)]
elif isinstance(content, str):
result = [self._convert_content_str_to_input_text(content, role)]
verbose_logger.debug(f"Chat provider: String content -> {result}")
verbose_logger.debug("Chat provider: String content -> %s", result)
return result
elif isinstance(content, list):
result = []
for i, item in enumerate(content):
verbose_logger.debug(f"Chat provider: Processing content item {i}: {type(item)} = {item}")
verbose_logger.debug("Chat provider: Processing content item %s: %s = %s", i, type(item), item)
if isinstance(item, str):
converted = self._convert_content_str_to_input_text(item, role)
result.append(converted)
verbose_logger.debug(f"Chat provider: -> {converted}")
verbose_logger.debug("Chat provider: -> %s", converted)
elif isinstance(item, dict):
# Handle multimodal content
original_type = item.get("type")
if original_type == "text":
converted = self._convert_content_str_to_input_text(item.get("text", ""), role)
result.append(converted)
verbose_logger.debug(f"Chat provider: text -> {converted}")
verbose_logger.debug("Chat provider: text -> %s", converted)
elif original_type == "image_url":
# Map to responses API image format
converted = cast(
@ -875,14 +875,14 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
),
)
result.append(converted)
verbose_logger.debug(f"Chat provider: image_url -> {converted}")
verbose_logger.debug("Chat provider: image_url -> %s", converted)
else:
# Try to map other types to responses API format
item_type = original_type or "input_text"
if item_type == "image":
converted = {"type": "input_image", **item}
result.append(converted)
verbose_logger.debug(f"Chat provider: image -> {converted}")
verbose_logger.debug("Chat provider: image -> %s", converted)
elif item_type == "file":
# Map Chat Completion file to Responses API input_file
# {"type": "file", "file": {"file_data": "...", "filename": "..."}}
@ -894,7 +894,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
if key in file_data:
converted[key] = file_data[key]
result.append(converted)
verbose_logger.debug(f"Chat provider: file -> {converted}")
verbose_logger.debug("Chat provider: file -> %s", converted)
elif item_type in [
"input_text",
"input_image",
@ -906,17 +906,17 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
]:
# Already in responses API format
result.append(item)
verbose_logger.debug(f"Chat provider: passthrough -> {item}")
verbose_logger.debug("Chat provider: passthrough -> %s", item)
else:
# Default to input_text for unknown types
converted = self._convert_content_str_to_input_text(str(item.get("text", item)), role)
result.append(converted)
verbose_logger.debug(f"Chat provider: unknown({original_type}) -> {converted}")
verbose_logger.debug(f"Chat provider: Final converted content: {result}")
verbose_logger.debug("Chat provider: unknown(%s) -> %s", original_type, converted)
verbose_logger.debug("Chat provider: Final converted content: %s", result)
return result
else:
result = [self._convert_content_str_to_input_text(str(content), role)]
verbose_logger.debug(f"Chat provider: Other content type -> {result}")
verbose_logger.debug("Chat provider: Other content type -> %s", result)
return result
def _convert_tools_to_responses_format(self, tools: list[dict[str, Any]]) -> list["ALL_RESPONSES_API_TOOL_PARAMS"]:
@ -1111,13 +1111,13 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
annotation_dict = annotation
else:
# Skip unsupported annotation types
verbose_logger.debug(f"Skipping unsupported annotation type: {type(annotation)}")
verbose_logger.debug("Skipping unsupported annotation type: %s", type(annotation))
continue
result.append(annotation_dict) # type: ignore
except Exception as e:
# Skip malformed annotations
verbose_logger.debug(f"Skipping malformed annotation: {annotation}, error: {e}")
verbose_logger.debug("Skipping malformed annotation: %s, error: %s", annotation, e)
continue
return result if result else None
@ -1222,11 +1222,11 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
):
return ModelResponseStream(**parsed_chunk)
verbose_logger.debug(f"Chat provider: Processing event type: {event_type}")
verbose_logger.debug("Chat provider: Processing event type: %s", event_type)
if event_type == "response.created":
# Initial response creation event
verbose_logger.debug(f"Chat provider: response.created -> {parsed_chunk}")
verbose_logger.debug("Chat provider: response.created -> %s", parsed_chunk)
return ModelResponseStream(
choices=[
StreamingChoices(
@ -1434,7 +1434,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
else:
pass
# For any unhandled event types, create a minimal valid chunk or skip
verbose_logger.debug(f"Chat provider: Unhandled event type '{event_type}', creating empty chunk")
verbose_logger.debug("Chat provider: Unhandled event type '%s', creating empty chunk", event_type)
# Return a minimal valid chunk for unknown events
return ModelResponseStream(
@ -1457,7 +1457,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
Returns:
ModelResponseStream: OpenAI-formatted streaming chunk
"""
verbose_logger.debug(f"Chat provider: transform_streaming_response called with chunk: {chunk}")
verbose_logger.debug("Chat provider: transform_streaming_response called with chunk: %s", chunk)
return self._with_stream_scoped_id(
OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
chunk, tool_call_index_map=self._tool_call_index_map

View file

@ -197,6 +197,16 @@ RUNWAYML_POLLING_TIMEOUT = int(os.getenv("RUNWAYML_POLLING_TIMEOUT", 600)) # 10
########## Networking constants ##############################################################
_DEFAULT_TTL_FOR_HTTPX_CLIENTS = 3600 # 1 hour, re-use the same httpx client for 1 hour
# The earliest an evicted, litellm-created client may be closed. A request handed the
# client just before eviction is still using it, so nothing is closed inside this window;
# past it, the client is closed once it reports no connection in flight.
EVICTED_LLM_CLIENT_CLOSE_GRACE_SECONDS = 900
# How many evicted clients may be queued for closing at once. Past this, an evicted client
# is left to the collector rather than letting a cache-churning workload grow the queue
# without bound. Each queued entry is ~100 bytes and comes due within one grace window.
EVICTED_LLM_CLIENT_CLOSE_MAX_PENDING = 10_000
# Aiohttp connection pooling - prevents memory leaks from unbounded connection growth
# Set to 0 for unlimited (not recommended for production)
AIOHTTP_CONNECTOR_LIMIT = int(os.getenv("AIOHTTP_CONNECTOR_LIMIT", 1000))
@ -303,6 +313,7 @@ MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES = int(os.getenv("MAX_LONG_SIDE_FOR_IMAGE_HIGH_R
MAX_TILE_WIDTH = int(os.getenv("MAX_TILE_WIDTH", 512))
MAX_TILE_HEIGHT = int(os.getenv("MAX_TILE_HEIGHT", 512))
OPENAI_FILE_SEARCH_COST_PER_1K_CALLS = float(os.getenv("OPENAI_FILE_SEARCH_COST_PER_1K_CALLS", 2.5 / 1000))
GROQ_BROWSER_VISIT_WEBSITE_COST_PER_CALL = 1.0 / 1000
# Azure OpenAI Assistants feature costs
# Source: https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/
AZURE_FILE_SEARCH_COST_PER_GB_PER_DAY = float(

View file

@ -273,7 +273,7 @@ def _get_additional_costs(
completion_tokens=completion_tokens,
)
except Exception as e:
verbose_logger.debug(f"Error calculating additional costs: {e}")
verbose_logger.debug("Error calculating additional costs: %s", e)
return None
@ -715,7 +715,7 @@ def _get_provider_for_cost_calc(
_, custom_llm_provider, _, _ = litellm.get_llm_provider(model=model)
except Exception as e:
verbose_logger.debug(
f"litellm.cost_calculator.py::_get_provider_for_cost_calc() - Error inferring custom_llm_provider - {e!s}"
"litellm.cost_calculator.py::_get_provider_for_cost_calc() - Error inferring custom_llm_provider - %s", e
)
return None
@ -896,7 +896,7 @@ def _get_usage_object(
elif isinstance(usage_obj, BaseModel):
return Usage(**usage_obj.model_dump())
else:
verbose_logger.debug(f"Unknown usage object type: {type(usage_obj)}, usage_obj: {usage_obj}")
verbose_logger.debug("Unknown usage object type: %s, usage_obj: %s", type(usage_obj), usage_obj)
return None
@ -994,16 +994,17 @@ def _apply_cost_margin(
if custom_llm_provider and custom_llm_provider in litellm.cost_margin_config:
margin_config = litellm.cost_margin_config[custom_llm_provider]
if verbose_logger.isEnabledFor(logging.DEBUG):
verbose_logger.debug(f"Found provider-specific margin config for {custom_llm_provider}: {margin_config}")
verbose_logger.debug("Found provider-specific margin config for %s: %s", custom_llm_provider, margin_config)
elif "global" in litellm.cost_margin_config:
margin_config = litellm.cost_margin_config["global"]
if verbose_logger.isEnabledFor(logging.DEBUG):
verbose_logger.debug(f"Using global margin config: {margin_config}")
verbose_logger.debug("Using global margin config: %s", margin_config)
else:
if verbose_logger.isEnabledFor(logging.DEBUG):
verbose_logger.debug(
f"No margin config found. Provider: {custom_llm_provider}, "
f"Available configs: {list(litellm.cost_margin_config.keys())}"
"No margin config found. Provider: %s, Available configs: %s",
custom_llm_provider,
list(litellm.cost_margin_config.keys()),
)
if margin_config is not None:
@ -1051,6 +1052,8 @@ def _store_cost_breakdown_in_logging_obj(
cache_read_cost: float | None = None,
cache_creation_cost: float | None = None,
reasoning_cost: float | None = None,
service_tier: str | None = None,
data_residency: str | None = None,
) -> None:
"""
Helper function to store cost breakdown in the logging object.
@ -1068,6 +1071,8 @@ def _store_cost_breakdown_in_logging_obj(
margin_percent: Margin percentage applied (0.10 = 10%)
margin_fixed_amount: Fixed margin amount in USD
margin_total_amount: Total margin added in USD
service_tier: Tier the costs above were priced on, already resolved
data_residency: Region uplift the costs above were priced on, already resolved
"""
if litellm_logging_obj is None:
return
@ -1089,10 +1094,12 @@ def _store_cost_breakdown_in_logging_obj(
cache_read_cost=cache_read_cost,
cache_creation_cost=cache_creation_cost,
reasoning_cost=reasoning_cost,
service_tier=service_tier,
data_residency=data_residency,
)
except Exception as breakdown_error:
verbose_logger.debug(f"Error storing cost breakdown: {breakdown_error!s}")
verbose_logger.debug("Error storing cost breakdown: %s", breakdown_error)
# Don't fail the main cost calculation if breakdown storage fails
@ -1219,7 +1226,7 @@ def completion_cost(
for idx, model in enumerate(potential_model_names):
try:
if verbose_logger.isEnabledFor(logging.DEBUG):
verbose_logger.debug(f"selected model name for cost calculation: {model}")
verbose_logger.debug("selected model name for cost calculation: %s", model)
if completion_response is not None and (
isinstance(completion_response, BaseModel) or isinstance(completion_response, dict)
@ -1315,7 +1322,8 @@ def completion_cost(
) # strip the llm provider from the model name -> for image gen cost calculation
except Exception as e:
verbose_logger.debug(
f"litellm.cost_calculator.py::completion_cost() - Error inferring custom_llm_provider - {e!s}"
"litellm.cost_calculator.py::completion_cost() - Error inferring custom_llm_provider - %s",
e,
)
if CostCalculatorUtils._call_type_has_image_response(call_type) and isinstance(
completion_response, ImageResponse
@ -1469,6 +1477,8 @@ def completion_cost(
margin_percent=margin_percent,
margin_fixed_amount=margin_fixed_amount,
margin_total_amount=margin_total_amount,
service_tier=service_tier,
data_residency=data_residency,
)
return _final_cost
@ -1657,12 +1667,14 @@ def completion_cost(
cache_read_cost=_cache_read_cost,
cache_creation_cost=_cache_creation_cost,
reasoning_cost=_reasoning_cost,
service_tier=service_tier,
data_residency=data_residency,
)
return _final_cost
except Exception as e:
verbose_logger.debug(
f"litellm.cost_calculator.py::completion_cost() - Error calculating cost for model={model} - {e!s}"
"litellm.cost_calculator.py::completion_cost() - Error calculating cost for model=%s - %s", model, e
)
if idx == len(potential_model_names) - 1:
raise e
@ -1878,7 +1890,7 @@ def vector_store_search_cost(
)
if config is None:
verbose_logger.debug(f"Vector store search is not supported for {custom_llm_provider}")
verbose_logger.debug("Vector store search is not supported for %s", custom_llm_provider)
return 0.0, 0.0
return config.calculate_vector_store_cost(
@ -1966,7 +1978,7 @@ def default_image_cost_calculator(
# gpt-image-1 models use low, medium, high quality. If user did not specify quality, use medium fot gpt-image-1 model family
model_name_with_v2_quality = f"{ImageGenerationRequestQuality.HIGH.value}/{base_model_name}"
verbose_logger.debug(f"Looking up cost for models: {model_name_with_quality}, {base_model_name}")
verbose_logger.debug("Looking up cost for models: %s, %s", model_name_with_quality, base_model_name)
model_without_provider = f"{size_str}/{model.split('/')[-1]}"
model_with_quality_without_provider = f"{quality}/{model_without_provider}" if quality else model_without_provider
@ -2036,7 +2048,7 @@ def default_video_cost_calculator(
model_name_without_custom_llm_provider = model.replace(f"{custom_llm_provider}/", "")
base_model_name = f"{custom_llm_provider}/{model_name_without_custom_llm_provider}"
verbose_logger.debug(f"Looking up cost for video model: {base_model_name}")
verbose_logger.debug("Looking up cost for video model: %s", base_model_name)
model_without_provider = model.split("/")[-1]
@ -2072,7 +2084,8 @@ def default_video_cost_calculator(
# If no cost information found, return 0
verbose_logger.info(
f"No cost information found for video model {model}. Please add pricing to model_prices_and_context_window.json"
"No cost information found for video model %s. Please add pricing to model_prices_and_context_window.json",
model,
)
return 0.0
@ -2351,6 +2364,7 @@ def handle_realtime_stream_cost_calculation(
cost_for_built_in_tools_cost_usd_dollar=0.0,
total_cost_usd_dollar=total_cost,
additional_costs={"transcription_cost": transcription_cost} if transcription_cost > 0 else None,
data_residency=data_residency,
)
return total_cost

View file

@ -1140,7 +1140,7 @@ class MidStreamFallbackError(ServiceUnavailableError): # type: ignore
if self.max_retries:
_message += f", LiteLLM Max Retries: {self.max_retries}"
if self.original_exception:
_message += f" Original exception: {type(self.original_exception).__name__}: {self.original_exception!s}"
_message += f" Original exception: {type(self.original_exception).__name__}: {self.original_exception}"
return _message
def __repr__(self):

View file

@ -364,7 +364,7 @@ class MCPClient:
try:
await session_ctx.__aexit__(None, None, None)
except BaseException as e:
verbose_logger.debug(f"Error during session context exit: {e}")
verbose_logger.debug("Error during session context exit: %s", e)
except BaseException as e:
in_flight_error = e
raise
@ -372,7 +372,7 @@ class MCPClient:
try:
await transport_ctx.__aexit__(None, None, None)
except BaseException as exit_error:
verbose_logger.debug(f"Error during transport context exit: {exit_error}")
verbose_logger.debug("Error during transport context exit: %s", exit_error)
root_cause = _first_non_cancelled_cause(exit_error)
if root_cause is not None and isinstance(in_flight_error, asyncio.CancelledError):
raise root_cause from in_flight_error
@ -402,7 +402,7 @@ class MCPClient:
try:
await http_client.aclose()
except BaseException as e:
verbose_logger.debug(f"Error during http_client cleanup: {e}")
verbose_logger.debug("Error during http_client cleanup: %s", e)
def update_auth_value(self, mcp_auth_value: str | dict[str, str]):
"""
@ -464,7 +464,7 @@ class MCPClient:
"""Create an httpx.AsyncClient with LiteLLM's SSL configuration."""
# Get unified SSL configuration using the same logic as http_handler.py
ssl_config = get_ssl_configuration(self.ssl_verify)
verbose_logger.debug(f"MCP client using SSL configuration: {type(ssl_config).__name__}")
verbose_logger.debug("MCP client using SSL configuration: %s", type(ssl_config).__name__)
# The MCP SDK's sse_client and streamable_http_client call this factory without
# passing auth=, so the fallback is used: a v2-resolved auth if present, else the
# SigV4 aws_auth. Both are None for the common case — no behavior change.
@ -490,7 +490,7 @@ class MCPClient:
MCP client (triggering the upstream OAuth flow) rather than
masking them as "connected, no tools".
"""
verbose_logger.debug(f"MCP client listing tools from {self.server_url or 'stdio'}")
verbose_logger.debug("MCP client listing tools from %s", self.server_url or "stdio")
async def _list_tools_operation(session: ClientSession):
return await session.list_tools()
@ -499,7 +499,9 @@ class MCPClient:
result = await self.run_with_session(_list_tools_operation, quiet_on_error=raise_on_error)
tool_count = len(result.tools)
tool_names = [tool.name for tool in result.tools]
verbose_logger.info(f"MCP client listed {tool_count} tools from {self.server_url or 'stdio'}: {tool_names}")
verbose_logger.info(
"MCP client listed %s tools from %s: %s", tool_count, self.server_url or "stdio", tool_names
)
return result.tools
except asyncio.CancelledError:
verbose_logger.warning("MCP client list_tools was cancelled")
@ -515,7 +517,7 @@ class MCPClient:
_log(
f"MCP client list_tools failed - "
f"Error Type: {error_type}, "
f"Error: {e!s}, "
f"Error: {e}, "
f"Server: {self.server_url or 'stdio'}, "
f"Transport: {self.transport_type}"
)
@ -536,7 +538,7 @@ class MCPClient:
def error_tool_result(exc: Exception) -> MCPCallToolResult:
"""The error result ``call_tool`` returns when it swallows a failure (no re-execution)."""
return MCPCallToolResult(
content=[TextContent(type="text", text=f"{type(exc).__name__}: {exc!s}")],
content=[TextContent(type="text", text=f"{type(exc).__name__}: {exc}")],
isError=True,
)
@ -555,7 +557,7 @@ class MCPClient:
an upstream 401 so it can re-mint the exchanged token and retry once; every other
caller keeps the default and gets graceful ``isError`` degradation.
"""
verbose_logger.info(f"MCP client calling tool '{call_tool_request_params.name}'")
verbose_logger.info("MCP client calling tool '%s'", call_tool_request_params.name)
async def on_progress(progress: float, total: float | None, message: str | None):
percentage = (progress / total * 100) if total else 0
@ -568,7 +570,7 @@ class MCPClient:
try:
await host_progress_callback(progress, total)
except Exception as e:
verbose_logger.warning(f"Failed to forward to Host: {e}")
verbose_logger.warning("Failed to forward to Host: %s", e)
async def _call_tool_operation(session: ClientSession):
verbose_logger.debug("MCP client sending tool call to session")
@ -580,16 +582,16 @@ class MCPClient:
try:
tool_result = await self.run_with_session(_call_tool_operation, quiet_on_error=raise_on_error)
verbose_logger.info(f"MCP client tool call '{call_tool_request_params.name}' completed successfully")
verbose_logger.info("MCP client tool call '%s' completed successfully", call_tool_request_params.name)
return tool_result
except asyncio.CancelledError:
verbose_logger.warning(f"MCP client tool call timed out after {self.timeout}s for {self.server_url}")
verbose_logger.warning("MCP client tool call timed out after %ss for %s", self.timeout, self.server_url)
raise
except Exception as e:
import traceback
error_trace = traceback.format_exc()
verbose_logger.debug(f"MCP client tool call traceback:\n{error_trace}")
verbose_logger.debug("MCP client tool call traceback:\n%s", error_trace)
# Log detailed error information
error_type = type(e).__name__
# When the caller opted into raise_on_error it owns the exception and logs it at the
@ -601,7 +603,7 @@ class MCPClient:
_log(
f"MCP client call_tool failed - "
f"Error Type: {error_type}, "
f"Error: {e!s}, "
f"Error: {e}, "
f"Tool: {call_tool_request_params.name}, "
f"Server: {self.server_url or 'stdio'}, "
f"Transport: {self.transport_type}"
@ -619,7 +621,7 @@ class MCPClient:
async def list_prompts(self) -> list[Prompt]:
"""List available prompts from the server."""
verbose_logger.debug(f"MCP client listing tools from {self.server_url or 'stdio'}")
verbose_logger.debug("MCP client listing tools from %s", self.server_url or "stdio")
async def _list_prompts_operation(session: ClientSession):
return await session.list_prompts()
@ -629,7 +631,7 @@ class MCPClient:
prompt_count = len(result.prompts)
prompt_names = [prompt.name for prompt in result.prompts]
verbose_logger.info(
f"MCP client listed {prompt_count} tools from {self.server_url or 'stdio'}: {prompt_names}"
"MCP client listed %s tools from %s: %s", prompt_count, self.server_url or "stdio", prompt_names
)
return result.prompts
except asyncio.CancelledError:
@ -638,11 +640,11 @@ class MCPClient:
except Exception as e:
error_type = type(e).__name__
verbose_logger.error(
f"MCP client list_prompts failed - "
f"Error Type: {error_type}, "
f"Error: {e!s}, "
f"Server: {self.server_url or 'stdio'}, "
f"Transport: {self.transport_type}"
"MCP client list_prompts failed - Error Type: %s, Error: %s, Server: %s, Transport: %s",
error_type,
e,
self.server_url or "stdio",
self.transport_type,
)
# Check if it's a stream/connection error
if "BrokenResourceError" in error_type or "Broken" in error_type:
@ -655,7 +657,7 @@ class MCPClient:
async def get_prompt(self, get_prompt_request_params: GetPromptRequestParams) -> GetPromptResult:
"""Fetch a prompt definition from the MCP server."""
verbose_logger.info(f"MCP client fetching prompt '{get_prompt_request_params.name}'")
verbose_logger.info("MCP client fetching prompt '%s'", get_prompt_request_params.name)
async def _get_prompt_operation(session: ClientSession):
verbose_logger.debug("MCP client sending get_prompt request to session")
@ -666,7 +668,7 @@ class MCPClient:
try:
get_prompt_result = await self.run_with_session(_get_prompt_operation)
verbose_logger.info(f"MCP client get_prompt '{get_prompt_request_params.name}' completed successfully")
verbose_logger.info("MCP client get_prompt '%s' completed successfully", get_prompt_request_params.name)
return get_prompt_result
except asyncio.CancelledError:
verbose_logger.warning("MCP client get_prompt was cancelled")
@ -675,16 +677,16 @@ class MCPClient:
import traceback
error_trace = traceback.format_exc()
verbose_logger.debug(f"MCP client get_prompt traceback:\n{error_trace}")
verbose_logger.debug("MCP client get_prompt traceback:\n%s", error_trace)
# Log detailed error information
error_type = type(e).__name__
verbose_logger.error(
f"MCP client get_prompt failed - "
f"Error Type: {error_type}, "
f"Error: {e!s}, "
f"Prompt: {get_prompt_request_params.name}, "
f"Server: {self.server_url or 'stdio'}, "
f"Transport: {self.transport_type}"
"MCP client get_prompt failed - Error Type: %s, Error: %s, Prompt: %s, Server: %s, Transport: %s",
error_type,
e,
get_prompt_request_params.name,
self.server_url or "stdio",
self.transport_type,
)
# Check if it's a stream/connection error
if "BrokenResourceError" in error_type or "Broken" in error_type:
@ -696,7 +698,7 @@ class MCPClient:
async def list_resources(self) -> list[Resource]:
"""List available resources from the server."""
verbose_logger.debug(f"MCP client listing resources from {self.server_url or 'stdio'}")
verbose_logger.debug("MCP client listing resources from %s", self.server_url or "stdio")
async def _list_resources_operation(session: ClientSession):
return await session.list_resources()
@ -706,7 +708,7 @@ class MCPClient:
resource_count = len(result.resources)
resource_names = [resource.name for resource in result.resources]
verbose_logger.info(
f"MCP client listed {resource_count} resources from {self.server_url or 'stdio'}: {resource_names}"
"MCP client listed %s resources from %s: %s", resource_count, self.server_url or "stdio", resource_names
)
return result.resources
except asyncio.CancelledError:
@ -715,11 +717,11 @@ class MCPClient:
except Exception as e:
error_type = type(e).__name__
verbose_logger.error(
f"MCP client list_resources failed - "
f"Error Type: {error_type}, "
f"Error: {e!s}, "
f"Server: {self.server_url or 'stdio'}, "
f"Transport: {self.transport_type}"
"MCP client list_resources failed - Error Type: %s, Error: %s, Server: %s, Transport: %s",
error_type,
e,
self.server_url or "stdio",
self.transport_type,
)
# Check if it's a stream/connection error
if "BrokenResourceError" in error_type or "Broken" in error_type:
@ -732,7 +734,7 @@ class MCPClient:
async def list_resource_templates(self) -> list[ResourceTemplate]:
"""List available resource templates from the server."""
verbose_logger.debug(f"MCP client listing resource templates from {self.server_url or 'stdio'}")
verbose_logger.debug("MCP client listing resource templates from %s", self.server_url or "stdio")
async def _list_resource_templates_operation(session: ClientSession):
return await session.list_resource_templates()
@ -742,7 +744,10 @@ class MCPClient:
resource_template_count = len(result.resourceTemplates)
resource_template_names = [resourceTemplate.name for resourceTemplate in result.resourceTemplates]
verbose_logger.info(
f"MCP client listed {resource_template_count} resource templates from {self.server_url or 'stdio'}: {resource_template_names}"
"MCP client listed %s resource templates from %s: %s",
resource_template_count,
self.server_url or "stdio",
resource_template_names,
)
return result.resourceTemplates
except asyncio.CancelledError:
@ -751,11 +756,11 @@ class MCPClient:
except Exception as e:
error_type = type(e).__name__
verbose_logger.error(
f"MCP client list_resource_templates failed - "
f"Error Type: {error_type}, "
f"Error: {e!s}, "
f"Server: {self.server_url or 'stdio'}, "
f"Transport: {self.transport_type}"
"MCP client list_resource_templates failed - Error Type: %s, Error: %s, Server: %s, Transport: %s",
error_type,
e,
self.server_url or "stdio",
self.transport_type,
)
# Check if it's a stream/connection error
if "BrokenResourceError" in error_type or "Broken" in error_type:
@ -768,7 +773,7 @@ class MCPClient:
async def read_resource(self, url: AnyUrl) -> ReadResourceResult:
"""Fetch resource contents from the MCP server."""
verbose_logger.info(f"MCP client fetching resource '{url}'")
verbose_logger.info("MCP client fetching resource '%s'", url)
async def _read_resource_operation(session: ClientSession):
verbose_logger.debug("MCP client sending read_resource request to session")
@ -776,7 +781,7 @@ class MCPClient:
try:
read_resource_result = await self.run_with_session(_read_resource_operation)
verbose_logger.info(f"MCP client read_resource '{url}' completed successfully")
verbose_logger.info("MCP client read_resource '%s' completed successfully", url)
return read_resource_result
except asyncio.CancelledError:
verbose_logger.warning("MCP client read_resource was cancelled")
@ -785,16 +790,16 @@ class MCPClient:
import traceback
error_trace = traceback.format_exc()
verbose_logger.debug(f"MCP client read_resource traceback:\n{error_trace}")
verbose_logger.debug("MCP client read_resource traceback:\n%s", error_trace)
# Log detailed error information
error_type = type(e).__name__
verbose_logger.error(
f"MCP client read_resource failed - "
f"Error Type: {error_type}, "
f"Error: {e!s}, "
f"Url: {url}, "
f"Server: {self.server_url or 'stdio'}, "
f"Transport: {self.transport_type}"
"MCP client read_resource failed - Error Type: %s, Error: %s, Url: %s, Server: %s, Transport: %s",
error_type,
e,
url,
self.server_url or "stdio",
self.transport_type,
)
# Check if it's a stream/connection error
if "BrokenResourceError" in error_type or "Broken" in error_type:

View file

@ -98,7 +98,7 @@ class GenerateContentToCompletionHandler:
return generate_content_response
except Exception as e:
raise ValueError(f"Error calling litellm.acompletion for generate_content: {e!s}")
raise ValueError(f"Error calling litellm.acompletion for generate_content: {e}")
@staticmethod
def generate_content_handler(
@ -159,4 +159,4 @@ class GenerateContentToCompletionHandler:
return generate_content_response
except Exception as e:
raise ValueError(f"Error calling litellm.completion for generate_content: {e!s}")
raise ValueError(f"Error calling litellm.completion for generate_content: {e}")

View file

@ -104,9 +104,10 @@ class GoogleGenAIStreamWrapper(AdapterCompletionStreamWrapper):
except json.JSONDecodeError:
# This can happen if the stream is abruptly cut off mid-argument string.
verbose_logger.warning(
f"Could not parse tool call arguments at end of stream for index {tool_call_index}. "
f"Name: {tool_call_data['name']}. "
f"Partial args: {tool_call_data['arguments']}"
"Could not parse tool call arguments at end of stream for index %s. Name: %s. Partial args: %s",
tool_call_index,
tool_call_data["name"],
tool_call_data["arguments"],
)
if parts:
final_chunk = {
@ -662,7 +663,7 @@ class GoogleGenAIAdapter:
# Optimization: Skip chunks that have no new data
if not function_name and not args_chunk:
verbose_logger.debug(f"Skipping empty tool call chunk for index: {tool_call_index}")
verbose_logger.debug("Skipping empty tool call chunk for index: %s", tool_call_index)
continue
if function_name:

View file

@ -68,8 +68,8 @@ async def send_to_webhook(slackAlertingInstance: SlackAlertingType, item, count)
data=json.dumps(payload),
)
if response.status_code != 200:
verbose_proxy_logger.debug(f"Error sending slack alert to url={item['url']}. Error={response.text}")
verbose_proxy_logger.debug("Error sending slack alert to url=%s. Error=%s", item["url"], response.text)
except Exception as e:
verbose_proxy_logger.debug(f"Error sending slack alert: {e!s}")
verbose_proxy_logger.debug("Error sending slack alert: %s", e)
finally:
_print_alerting_payload_warning(payload, slackAlertingInstance=slackAlertingInstance)

View file

@ -1467,7 +1467,7 @@ Model Info:
try:
await self._flush_digest_buckets()
except Exception as e:
verbose_proxy_logger.debug(f"Error flushing digest buckets: {e!s}")
verbose_proxy_logger.debug("Error flushing digest buckets: %s", e)
await self.flush_queue()
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
@ -1502,7 +1502,7 @@ Model Info:
)
except Exception as e:
verbose_proxy_logger.error(
f"[Non-Blocking Error] Slack Alerting: Got error in logging LLM deployment latency: {e!s}"
"[Non-Blocking Error] Slack Alerting: Got error in logging LLM deployment latency: %s", e
)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
@ -1522,7 +1522,7 @@ Model Info:
)
)
except Exception as e:
verbose_logger.debug(f"Exception raises -{e!s}")
verbose_logger.debug("Exception raises -%s", e)
if isinstance(kwargs.get("exception", ""), APIError):
if "outage_alerts" in self.alert_types:
@ -1662,9 +1662,9 @@ Model Info:
)
except ValueError as ve:
verbose_proxy_logger.error(f"Invalid time range format: {ve}")
verbose_proxy_logger.error("Invalid time range format: %s", ve)
except Exception as e:
verbose_proxy_logger.error(f"Error sending spend report: {e}")
verbose_proxy_logger.error("Error sending spend report: %s", e)
async def send_monthly_spend_report(self):
""" """

View file

@ -143,8 +143,8 @@ class AnthropicCacheControlHook(CustomPromptManagement):
if limit_reached:
verbose_logger.warning(
f"AnthropicCacheControlHook: Reached the Anthropic limit of "
f"{MAX_CACHE_CONTROL_BLOCKS} cache_control blocks. Skipping further injection."
"AnthropicCacheControlHook: Reached the Anthropic limit of %s cache_control blocks. Skipping further injection.",
MAX_CACHE_CONTROL_BLOCKS,
)
return messages
@ -174,8 +174,10 @@ class AnthropicCacheControlHook(CustomPromptManagement):
return [targetted_index]
verbose_logger.warning(
f"AnthropicCacheControlHook: Provided index {original_index} is out of bounds for message list of length {len(messages)}. "
f"Targeted index was {targetted_index}. Skipping cache control injection for this point."
"AnthropicCacheControlHook: Provided index %s is out of bounds for message list of length %s. Targeted index was %s. Skipping cache control injection for this point.",
original_index,
len(messages),
targetted_index,
)
return []

View file

@ -185,9 +185,9 @@ class ArgillaLogger(CustomBatchLogger):
)
if response.status_code >= 300:
verbose_logger.error(f"Argilla Error: {response.status_code} - {response.text}")
verbose_logger.error("Argilla Error: %s - %s", response.status_code, response.text)
else:
verbose_logger.debug(f"Batch of {len(self.log_queue)} runs successfully created")
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
self.log_queue.clear()
except Exception:
@ -204,7 +204,7 @@ class ArgillaLogger(CustomBatchLogger):
random_sample = random.random()
if random_sample > sampling_rate:
verbose_logger.info(
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
)
return # Skip logging
verbose_logger.debug(
@ -217,7 +217,7 @@ class ArgillaLogger(CustomBatchLogger):
return
self.log_queue.append(data)
verbose_logger.debug(f"Langsmith, event added to queue. Will flush in {self.flush_interval} seconds...")
verbose_logger.debug("Langsmith, event added to queue. Will flush in %s seconds...", self.flush_interval)
if len(self.log_queue) >= self.batch_size:
self._send_batch()
@ -231,7 +231,7 @@ class ArgillaLogger(CustomBatchLogger):
random_sample = random.random()
if random_sample > sampling_rate:
verbose_logger.info(
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
)
return # Skip logging
verbose_logger.debug(
@ -272,7 +272,7 @@ class ArgillaLogger(CustomBatchLogger):
random_sample = random.random()
if random_sample > sampling_rate:
verbose_logger.info(
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
)
return # Skip logging
verbose_logger.info("Langsmith Failure Event Logging!")
@ -325,7 +325,7 @@ class ArgillaLogger(CustomBatchLogger):
response.raise_for_status()
if response.status_code >= 300:
verbose_logger.error(f"Argilla Error: {response.status_code} - {response.text}")
verbose_logger.error("Argilla Error: %s - %s", response.status_code, response.text)
else:
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
except httpx.HTTPStatusError:

View file

@ -461,7 +461,7 @@ def set_attributes(span: "Span", kwargs, response_obj, attributes: type[BaseLLMO
_set_response_attributes(span=span, response_obj=response_obj_for_attrs)
except Exception as e:
verbose_logger.error(f"[Arize/Phoenix] Failed to set OpenInference span attributes: {e}")
verbose_logger.error("[Arize/Phoenix] Failed to set OpenInference span attributes: %s", e)
if hasattr(span, "record_exception"):
span.record_exception(e)

View file

@ -169,7 +169,7 @@ class ArizeLogger(OpenTelemetry):
except Exception as e:
return {
"status": "unhealthy",
"error_message": f"Arize health check failed: {e!s}",
"error_message": f"Arize health check failed: {e}",
}
def construct_dynamic_otel_headers(

View file

@ -425,7 +425,7 @@ class ArizePhoenixLogger(OpenTelemetry): # type: ignore
endpoint = "http://localhost:6006/v1/traces"
protocol = "otlp_http"
verbose_logger.debug(
f"No PHOENIX_COLLECTOR_ENDPOINT found, using default local Phoenix endpoint: {endpoint}"
"No PHOENIX_COLLECTOR_ENDPOINT found, using default local Phoenix endpoint: %s", endpoint
)
otlp_auth_headers = None

View file

@ -339,7 +339,7 @@ class ArizePhoenixPromptManager(CustomPromptManagement):
# Log error but don't fail the call
import litellm
litellm._logging.verbose_proxy_logger.error(f"Error in Arize Phoenix prompt pre_call_hook: {e}")
litellm._logging.verbose_proxy_logger.error("Error in Arize Phoenix prompt pre_call_hook: %s", e)
return messages, litellm_params
def get_available_prompts(self) -> list[str]:

View file

@ -203,7 +203,7 @@ class AzureSentinelLogger(CustomBatchLogger):
await self.async_send_batch()
except Exception as e:
verbose_logger.exception(f"Azure Sentinel Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Azure Sentinel Layer Error - %s\n%s", e, traceback.format_exc())
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
"""
@ -233,7 +233,7 @@ class AzureSentinelLogger(CustomBatchLogger):
await self.async_send_batch()
except Exception as e:
verbose_logger.exception(f"Azure Sentinel Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Azure Sentinel Layer Error - %s\n%s", e, traceback.format_exc())
async def async_log_audit_log_event(self, audit_log: StandardAuditLogPayload) -> None:
"""
@ -256,7 +256,7 @@ class AzureSentinelLogger(CustomBatchLogger):
await self.async_send_audit_batch()
except Exception as e:
verbose_logger.exception(f"Azure Sentinel Audit Log Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Azure Sentinel Audit Log Layer Error - %s\n%s", e, traceback.format_exc())
async def async_send_batch(self):
"""
@ -323,7 +323,7 @@ class AzureSentinelLogger(CustomBatchLogger):
)
except Exception as e:
verbose_logger.exception(f"Azure Sentinel Error sending batch API - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Azure Sentinel Error sending batch API - %s\n%s", e, traceback.format_exc())
finally:
log_queue.clear()

View file

@ -54,7 +54,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
super().__init__(**kwargs, flush_lock=self.flush_lock)
except Exception as e:
verbose_logger.exception(
f"AzureBlobStorageLogger: Got exception on init AzureBlobStorageLogger client {e!s}"
"AzureBlobStorageLogger: Got exception on init AzureBlobStorageLogger client %s", e
)
raise e
@ -79,7 +79,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
self.log_queue.append(standard_logging_payload)
except Exception as e:
verbose_logger.exception(f"AzureBlobStorageLogger Layer Error - {e!s}")
verbose_logger.exception("AzureBlobStorageLogger Layer Error - %s", e)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
"""
@ -101,7 +101,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
self.log_queue.append(standard_logging_payload)
except Exception as e:
verbose_logger.exception(f"AzureBlobStorageLogger Layer Error - {e!s}")
verbose_logger.exception("AzureBlobStorageLogger Layer Error - %s", e)
async def async_send_batch(self):
"""
@ -124,7 +124,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
await self.async_upload_payload_to_azure_blob_storage(payload=payload)
except Exception as e:
verbose_logger.exception(f"AzureBlobStorageLogger Error sending batch API - {e!s}")
verbose_logger.exception("AzureBlobStorageLogger Error sending batch API - %s", e)
async def async_upload_payload_to_azure_blob_storage(self, payload: StandardLoggingPayload):
"""
@ -150,16 +150,16 @@ class AzureBlobStorageLogger(CustomBatchLogger):
await self._append_data(async_client, base_url, json_payload)
await self._flush_data(async_client, base_url, len(payload_bytes))
verbose_logger.debug(f"Successfully uploaded log to Azure Blob Storage: {filename}")
verbose_logger.debug("Successfully uploaded log to Azure Blob Storage: %s", filename)
except Exception as e:
verbose_logger.exception(f"Error uploading to Azure Blob Storage: {e!s}")
verbose_logger.exception("Error uploading to Azure Blob Storage: %s", e)
raise e
async def _create_file(self, client: AsyncHTTPHandler, base_url: str):
"""Helper method to create the file resource"""
try:
verbose_logger.debug(f"Creating file resource at: {base_url}")
verbose_logger.debug("Creating file resource at: %s", base_url)
headers = {
"x-ms-version": AZURE_STORAGE_MSFT_VERSION,
"Content-Length": "0",
@ -169,13 +169,13 @@ class AzureBlobStorageLogger(CustomBatchLogger):
response.raise_for_status()
verbose_logger.debug("Successfully created file resource")
except Exception as e:
verbose_logger.exception(f"Error creating file resource: {e!s}")
verbose_logger.exception("Error creating file resource: %s", e)
raise
async def _append_data(self, client: AsyncHTTPHandler, base_url: str, json_payload: str):
"""Helper method to append data to the file"""
try:
verbose_logger.debug(f"Appending data to file: {base_url}")
verbose_logger.debug("Appending data to file: %s", base_url)
headers = {
"x-ms-version": AZURE_STORAGE_MSFT_VERSION,
"Content-Type": "application/json",
@ -189,13 +189,13 @@ class AzureBlobStorageLogger(CustomBatchLogger):
response.raise_for_status()
verbose_logger.debug("Successfully appended data")
except Exception as e:
verbose_logger.exception(f"Error appending data: {e!s}")
verbose_logger.exception("Error appending data: %s", e)
raise
async def _flush_data(self, client: AsyncHTTPHandler, base_url: str, position: int):
"""Helper method to flush the data"""
try:
verbose_logger.debug(f"Flushing data at position {position}")
verbose_logger.debug("Flushing data at position %s", position)
headers = {
"x-ms-version": AZURE_STORAGE_MSFT_VERSION,
"Content-Length": "0",
@ -205,7 +205,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
response.raise_for_status()
verbose_logger.debug("Successfully flushed data")
except Exception as e:
verbose_logger.exception(f"Error flushing data: {e!s}")
verbose_logger.exception("Error flushing data: %s", e)
raise
####### Helper methods to managing Authentication to Azure Storage #######
@ -229,7 +229,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
)
# Token typically expires in 1 hour
self.token_expiry = datetime.now() + timedelta(hours=1)
verbose_logger.debug(f"New token will expire at {self.token_expiry}")
verbose_logger.debug("New token will expire at %s", self.token_expiry)
def get_azure_ad_token_from_azure_storage(
self,
@ -324,7 +324,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
# check if the directory exists
if not await directory_client.exists():
await directory_client.create_directory()
verbose_logger.debug(f"Created directory: {today}")
verbose_logger.debug("Created directory: %s", today)
# Create a file client
file_name = f"{payload.get('id') or str(uuid.uuid4())}.json"
@ -342,7 +342,7 @@ class AzureBlobStorageLogger(CustomBatchLogger):
# Flush the content to finalize the file
await file_client.flush_data(position=len(content), offset=0)
verbose_logger.debug(f"Successfully uploaded and wrote to {today}/{file_name}")
verbose_logger.debug("Successfully uploaded and wrote to %s/%s", today, file_name)
except Exception as e:
verbose_logger.exception(f"Error occurred: {e!s}")
verbose_logger.exception("Error occurred: %s", e)

View file

@ -320,7 +320,7 @@ class BitBucketPromptManager(CustomPromptManagement):
# Log error but don't fail the call
import litellm
litellm._logging.verbose_proxy_logger.error(f"Error in BitBucket prompt pre_call_hook: {e}")
litellm._logging.verbose_proxy_logger.error("Error in BitBucket prompt pre_call_hook: %s", e)
return messages, litellm_params
def _parse_prompt_to_messages(self, prompt_content: str) -> list[AllMessageValues]:

View file

@ -89,7 +89,7 @@ def _mock_http_handler_post(
"""Monkey-patched HTTPHandler.post that intercepts Braintrust calls with endpoint-specific responses."""
# Only mock Braintrust API calls
if isinstance(url, str) and _is_braintrust_url(url):
verbose_logger.info(f"[BRAINTRUST MOCK] POST to {url}")
verbose_logger.info("[BRAINTRUST MOCK] POST to %s", url)
time.sleep(_MOCK_LATENCY_SECONDS)
# Return appropriate mock response based on endpoint
if "/project" in url:

View file

@ -38,7 +38,7 @@ class CloudZeroLogger(CustomLogger):
self.connection_id = connection_id or os.getenv("CLOUDZERO_CONNECTION_ID")
self.timezone = timezone or os.getenv("CLOUDZERO_TIMEZONE", "UTC")
verbose_logger.debug(
f"CloudZero Logger initialized with connection ID: {self.connection_id}, timezone: {self.timezone}"
"CloudZero Logger initialized with connection ID: %s, timezone: %s", self.connection_id, self.timezone
)
async def initialize_cloudzero_export_job(self):
@ -130,7 +130,7 @@ class CloudZeroLogger(CustomLogger):
verbose_logger.debug("CloudZero Logger: No usage data found to export")
return
verbose_logger.debug(f"CloudZero Logger: Processing {len(data)} records")
verbose_logger.debug("CloudZero Logger: Processing %s records", len(data))
# Transform data to CloudZero CBF format
transformer = CBFTransformer()
@ -147,13 +147,13 @@ class CloudZeroLogger(CustomLogger):
user_timezone=self.timezone,
)
verbose_logger.debug(f"CloudZero Logger: Transmitting {len(cbf_data)} records to CloudZero")
verbose_logger.debug("CloudZero Logger: Transmitting %s records to CloudZero", len(cbf_data))
streamer.send_batched(cbf_data, operation=operation)
verbose_logger.debug(f"CloudZero Logger: Successfully exported {len(cbf_data)} records to CloudZero")
verbose_logger.debug("CloudZero Logger: Successfully exported %s records to CloudZero", len(cbf_data))
except Exception as e:
verbose_logger.error(f"CloudZero Logger: Error exporting usage data: {e!s}")
verbose_logger.error("CloudZero Logger: Error exporting usage data: %s", e)
raise
async def dry_run_export_usage_data(self, limit: int | None = 10000):
@ -191,7 +191,7 @@ class CloudZeroLogger(CustomLogger):
},
}
verbose_logger.debug(f"CloudZero Dry Run: Processing {len(data)} records...")
verbose_logger.debug("CloudZero Dry Run: Processing %s records...", len(data))
# Convert usage data to dict format for response
usage_data_sample = data.head(50).to_dicts() # Return first 50 rows
@ -229,7 +229,7 @@ class CloudZeroLogger(CustomLogger):
)
total_tokens = sum(record.get("usage/amount", 0) for record in cbf_data_dict)
verbose_logger.debug(f"CloudZero Logger: Dry run completed for {len(cbf_data)} records")
verbose_logger.debug("CloudZero Logger: Dry run completed for %s records", len(cbf_data))
return {
"usage_data": usage_data_sample,
@ -244,8 +244,8 @@ class CloudZeroLogger(CustomLogger):
}
except Exception as e:
verbose_logger.error(f"CloudZero Logger: Error in dry run export: {e!s}")
verbose_logger.error(f"CloudZero Dry Run Error: {e!s}")
verbose_logger.error("CloudZero Logger: Error in dry run export: %s", e)
verbose_logger.error("CloudZero Dry Run Error: %s", e)
raise
def _display_cbf_data_on_screen(self, cbf_data):

View file

@ -98,4 +98,4 @@ class LiteLLMDatabase:
# This prevents schema mismatch errors when data types vary across rows
return pl.DataFrame(db_response, infer_schema_length=None)
except Exception as e:
raise Exception(f"Error retrieving usage data: {e!s}")
raise Exception(f"Error retrieving usage data: {e}")

View file

@ -47,7 +47,7 @@ class CustomBatchLogger(CustomLogger):
async def periodic_flush(self):
while True:
await asyncio.sleep(self.flush_interval)
verbose_logger.debug(f"CustomLogger periodic flush after {self.flush_interval} seconds")
verbose_logger.debug("CustomLogger periodic flush after %s seconds", self.flush_interval)
await self.flush_queue()
async def flush_queue(self):

View file

@ -864,7 +864,7 @@ class CustomGuardrail(CustomLogger):
if premium_user is not True:
verbose_logger.warning(
f"Trying to use premium guardrail without premium user {CommonProxyErrors.not_premium_user.value}"
"Trying to use premium guardrail without premium user %s", CommonProxyErrors.not_premium_user.value
)
return False
return True
@ -1028,7 +1028,7 @@ class CustomGuardrail(CustomLogger):
else:
guardrail_response = "allow"
verbose_logger.debug(f"Guardrail response: {response}")
verbose_logger.debug("Guardrail response: %s", response)
self.add_standard_logging_guardrail_information_to_request_data(
guardrail_json_response=guardrail_response,

View file

@ -915,19 +915,19 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
for callback_obj in all_callbacks:
if hasattr(callback_obj, "increment_callback_logging_failure"):
verbose_logger.debug(f"Incrementing callback failure metric for {callback_name}")
verbose_logger.debug("Incrementing callback failure metric for %s", callback_name)
callback_obj.increment_callback_logging_failure(callback_name=callback_name) # type: ignore
return
verbose_logger.debug(
f"No callback with increment_callback_logging_failure method found for {callback_name}. "
"Ensure 'prometheus' is in your callbacks config."
"No callback with increment_callback_logging_failure method found for %s. Ensure 'prometheus' is in your callbacks config.",
callback_name,
)
except Exception as e:
from litellm._logging import verbose_logger
verbose_logger.debug(f"Error in handle_callback_failure for {callback_name}: {e!s}")
verbose_logger.debug("Error in handle_callback_failure for %s: %s", callback_name, e)
async def _strip_base64_from_messages(
self,
@ -946,7 +946,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
"""
raw_messages: Any = payload.get("messages", [])
messages: list[Any] = raw_messages if isinstance(raw_messages, list) else []
verbose_logger.debug(f"[CustomLogger] Stripping base64 from {len(messages)} messages")
verbose_logger.debug("[CustomLogger] Stripping base64 from %s messages", len(messages))
if messages:
payload["messages"] = self._process_messages(messages=messages, max_depth=max_depth)
@ -958,7 +958,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
if isinstance(content, list):
total_items += len(content)
verbose_logger.debug(f"[CustomLogger] Completed base64 strip; retained {total_items} content items")
verbose_logger.debug("[CustomLogger] Completed base64 strip; retained %s content items", total_items)
return payload
def _strip_base64_from_messages_sync(
@ -978,7 +978,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
"""
raw_messages: Any = payload.get("messages", [])
messages: list[Any] = raw_messages if isinstance(raw_messages, list) else []
verbose_logger.debug(f"[CustomLogger] Stripping base64 from {len(messages)} messages")
verbose_logger.debug("[CustomLogger] Stripping base64 from %s messages", len(messages))
if messages:
payload["messages"] = self._process_messages(messages=messages, max_depth=max_depth)
@ -990,7 +990,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
if isinstance(content, list):
total_items += len(content)
verbose_logger.debug(f"[CustomLogger] Completed base64 strip; retained {total_items} content items")
verbose_logger.debug("[CustomLogger] Completed base64 strip; retained %s content items", total_items)
return payload
def _redact_base64(
@ -1001,12 +1001,12 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
) -> Any:
"""Recursively redact inline base64 from any nested structure with a max recursion depth limit."""
if depth > max_depth:
verbose_logger.warning(f"[CustomLogger] Max recursion depth {max_depth} reached while redacting base64")
verbose_logger.warning("[CustomLogger] Max recursion depth %s reached while redacting base64", max_depth)
return "[MAX_DEPTH_REACHED]"
if isinstance(value, str):
if _BASE64_INLINE_PATTERN.search(value):
verbose_logger.debug(f"[CustomLogger] Redacted inline base64 string: {value[:40]}...")
verbose_logger.debug("[CustomLogger] Redacted inline base64 string: %s...", value[:40])
return _BASE64_INLINE_PATTERN.sub("[BASE64_REDACTED]", value)
return value

View file

@ -237,7 +237,7 @@ class CustomSecretManager(BaseSecretManager):
Returns:
True if the secret manager is healthy, False otherwise
"""
verbose_logger.debug(f"Health check not implemented for {self.secret_manager_name}")
verbose_logger.debug("Health check not implemented for %s", self.secret_manager_name)
return True
def __repr__(self) -> str:

View file

@ -171,7 +171,7 @@ class DataDogLogger(
batch_size=_resolve_dd_batch_size(),
)
except Exception as e:
verbose_logger.exception(f"Datadog: Got exception on init Datadog client {e!s}")
verbose_logger.exception("Datadog: Got exception on init Datadog client %s", e)
raise e
def _get_datadog_params(self) -> dict:
@ -210,7 +210,7 @@ class DataDogLogger(
self.DD_API_KEY = dd_api_key or (
os.getenv("DD_API_KEY") if allow_env_credentials else None
) # Optional when using agent
verbose_logger.debug(f"Datadog: Using DD Agent at {self.intake_url}")
verbose_logger.debug("Datadog: Using DD Agent at %s", self.intake_url)
def _configure_dd_direct_api(
self,
@ -257,7 +257,7 @@ class DataDogLogger(
await self._log_async_event(kwargs, response_obj, start_time, end_time)
except Exception as e:
verbose_logger.exception(f"Datadog Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Datadog Layer Error - %s\n%s", e, traceback.format_exc())
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
try:
@ -265,7 +265,7 @@ class DataDogLogger(
await self._log_async_event(kwargs, response_obj, start_time, end_time)
except Exception as e:
verbose_logger.exception(f"Datadog Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Datadog Layer Error - %s\n%s", e, traceback.format_exc())
async def async_post_call_failure_hook(
self,
@ -340,7 +340,7 @@ class DataDogLogger(
if len(self.log_queue) >= self.batch_size:
await self.flush_queue()
except Exception as e:
verbose_logger.exception(f"Datadog: async_post_call_failure_hook - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Datadog: async_post_call_failure_hook - %s\n%s", e, traceback.format_exc())
return None
async def async_send_batch(self):
@ -376,11 +376,11 @@ class DataDogLogger(
self.log_queue = undelivered + self.log_queue
if self.is_mock_mode:
verbose_logger.debug(f"[DATADOG MOCK] Batch of {len(batch_to_send)} events successfully mocked")
verbose_logger.debug("[DATADOG MOCK] Batch of %s events successfully mocked", len(batch_to_send))
except Exception as e:
self.log_queue = batch_to_send + self.log_queue
verbose_logger.exception(f"Datadog Error sending batch API - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Datadog Error sending batch API - %s\n%s", e, traceback.format_exc())
async def _send_with_413_split(self, batch: list) -> list:
"""
@ -411,7 +411,7 @@ class DataDogLogger(
if isinstance(e, MaskedHTTPStatusError) and e.status_code == 413:
response = e.response
else:
verbose_logger.exception(f"Datadog Error sending batch API - {e!s}")
verbose_logger.exception("Datadog Error sending batch API - %s", e)
return self._undelivered(chunk, pending)
if response.status_code == 413:
@ -515,7 +515,7 @@ class DataDogLogger(
)
except Exception as e:
verbose_logger.exception(f"Datadog Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Datadog Layer Error - %s\n%s", e, traceback.format_exc())
async def _log_async_event(self, kwargs, response_obj, start_time, end_time):
dd_payload = self.create_datadog_logging_payload(
@ -526,7 +526,7 @@ class DataDogLogger(
)
self.log_queue.append(dd_payload)
verbose_logger.debug(f"Datadog, event added to queue. Will flush in {self.flush_interval} seconds...")
verbose_logger.debug("Datadog, event added to queue. Will flush in %s seconds...", self.flush_interval)
if len(self.log_queue) >= self.batch_size:
await self.flush_queue()
@ -653,7 +653,7 @@ class DataDogLogger(
self.log_queue.append(_dd_payload)
except Exception as e:
verbose_logger.exception(f"Datadog: Logger - Exception in async_service_failure_hook: {e}")
verbose_logger.exception("Datadog: Logger - Exception in async_service_failure_hook: %s", e)
async def async_service_success_hook(
self,
@ -692,7 +692,7 @@ class DataDogLogger(
self.log_queue.append(_dd_payload)
except Exception as e:
verbose_logger.exception(f"Datadog: Logger - Exception in async_service_failure_hook: {e}")
verbose_logger.exception("Datadog: Logger - Exception in async_service_failure_hook: %s", e)
def _create_v0_logging_payload(
self,

View file

@ -84,7 +84,7 @@ class DatadogCostManagementLogger(CustomBatchLogger):
await self.async_send_batch()
except Exception as e:
verbose_logger.exception(f"Datadog Cost Management: Error in async_log_success_event: {e!s}")
verbose_logger.exception("Datadog Cost Management: Error in async_log_success_event: %s", e)
async def async_send_batch(self):
if not self.log_queue:
@ -104,7 +104,7 @@ class DatadogCostManagementLogger(CustomBatchLogger):
await self._upload_to_datadog(aggregated_entries)
except Exception as e:
self.log_queue = batch_to_send + self.log_queue
verbose_logger.exception(f"Datadog Cost Management: Error in async_send_batch: {e!s}")
verbose_logger.exception("Datadog Cost Management: Error in async_send_batch: %s", e)
def _aggregate_costs(self, logs: list[StandardLoggingPayload]) -> list[DatadogFOCUSCostEntry]:
"""
@ -159,7 +159,7 @@ class DatadogCostManagementLogger(CustomBatchLogger):
aggregator[key]["BilledCost"] += cost
except Exception as e:
verbose_logger.warning(f"Error processing log for cost aggregation: {e}")
verbose_logger.warning("Error processing log for cost aggregation: %s", e)
continue
return list(aggregator.values())
@ -254,5 +254,5 @@ class DatadogCostManagementLogger(CustomBatchLogger):
response.raise_for_status()
verbose_logger.debug(
f"Datadog Cost Management: Uploaded {len(payload)} cost entries. Status: {response.status_code}"
"Datadog Cost Management: Uploaded %s cost entries. Status: %s", len(payload), response.status_code
)

View file

@ -89,7 +89,7 @@ class DataDogLLMObsLogger(CustomBatchLogger):
kwargs.update(dict_datadog_llm_obs_params)
CustomBatchLogger.__init__(self, **kwargs, flush_lock=self.flush_lock)
except Exception as e:
verbose_logger.exception(f"DataDogLLMObs: Error initializing - {e!s}")
verbose_logger.exception("DataDogLLMObs: Error initializing - %s", e)
raise e
def _configure_dd_agent(self, dd_agent_host: str):
@ -103,7 +103,7 @@ class DataDogLLMObsLogger(CustomBatchLogger):
agent_port = os.getenv("LITELLM_DD_LLM_OBS_PORT", "8126")
self.DD_SITE = "localhost" # Not used for URL construction in agent mode
self.intake_url = f"http://{dd_agent_host}:{agent_port}/api/intake/llm-obs/v1/trace/spans"
verbose_logger.debug(f"DataDogLLMObs: Using DD Agent at {self.intake_url}")
verbose_logger.debug("DataDogLLMObs: Using DD Agent at %s", self.intake_url)
def _configure_dd_direct_api(self):
"""
@ -137,34 +137,34 @@ class DataDogLLMObsLogger(CustomBatchLogger):
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
try:
verbose_logger.debug(f"DataDogLLMObs: Logging success event for model {kwargs.get('model', 'unknown')}")
verbose_logger.debug("DataDogLLMObs: Logging success event for model %s", kwargs.get("model", "unknown"))
payload = self.create_llm_obs_payload(kwargs, start_time, end_time)
verbose_logger.debug(f"DataDogLLMObs: Payload: {payload}")
verbose_logger.debug("DataDogLLMObs: Payload: %s", payload)
self.log_queue.append(payload)
if len(self.log_queue) >= self.batch_size:
await self.async_send_batch()
except Exception as e:
verbose_logger.exception(f"DataDogLLMObs: Error logging success event - {e!s}")
verbose_logger.exception("DataDogLLMObs: Error logging success event - %s", e)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
try:
verbose_logger.debug(f"DataDogLLMObs: Logging failure event for model {kwargs.get('model', 'unknown')}")
verbose_logger.debug("DataDogLLMObs: Logging failure event for model %s", kwargs.get("model", "unknown"))
payload = self.create_llm_obs_payload(kwargs, start_time, end_time)
verbose_logger.debug(f"DataDogLLMObs: Payload: {payload}")
verbose_logger.debug("DataDogLLMObs: Payload: %s", payload)
self.log_queue.append(payload)
if len(self.log_queue) >= self.batch_size:
await self.async_send_batch()
except Exception as e:
verbose_logger.exception(f"DataDogLLMObs: Error logging failure event - {e!s}")
verbose_logger.exception("DataDogLLMObs: Error logging failure event - %s", e)
async def async_send_batch(self):
try:
if not self.log_queue:
return
verbose_logger.debug(f"DataDogLLMObs: Flushing {len(self.log_queue)} events")
verbose_logger.debug("DataDogLLMObs: Flushing %s events", len(self.log_queue))
if self.is_mock_mode:
verbose_logger.debug("[DATADOG MOCK] Mock mode enabled - API calls will be intercepted")
@ -207,14 +207,14 @@ class DataDogLLMObsLogger(CustomBatchLogger):
)
if self.is_mock_mode:
verbose_logger.debug(f"[DATADOG MOCK] Batch of {len(self.log_queue)} events successfully mocked")
verbose_logger.debug("[DATADOG MOCK] Batch of %s events successfully mocked", len(self.log_queue))
else:
verbose_logger.debug(f"DataDogLLMObs: Successfully sent batch - status_code: {response.status_code}")
verbose_logger.debug("DataDogLLMObs: Successfully sent batch - status_code: %s", response.status_code)
self.log_queue.clear()
except httpx.HTTPStatusError as e:
verbose_logger.exception(f"DataDogLLMObs: Error sending batch - {e.response.text}")
verbose_logger.exception("DataDogLLMObs: Error sending batch - %s", e.response.text)
except Exception as e:
verbose_logger.exception(f"DataDogLLMObs: Error sending batch - {e!s}")
verbose_logger.exception("DataDogLLMObs: Error sending batch - %s", e)
def create_llm_obs_payload(self, kwargs: dict, start_time: datetime, end_time: datetime) -> LLMObsPayload:
standard_logging_payload: StandardLoggingPayload | None = kwargs.get("standard_logging_object")
@ -613,7 +613,7 @@ class DataDogLLMObsLogger(CustomBatchLogger):
try:
spend_metrics["user_api_key_spend"] = float(user_api_key_spend)
except (ValueError, TypeError):
verbose_logger.debug(f"Invalid user_api_key_spend value: {user_api_key_spend}")
verbose_logger.debug("Invalid user_api_key_spend value: %s", user_api_key_spend)
# API key budget reset datetime
user_api_key_budget_reset_at = metadata.get("user_api_key_budget_reset_at")
@ -640,10 +640,10 @@ class DataDogLLMObsLogger(CustomBatchLogger):
spend_metrics["user_api_key_budget_reset_at"] = iso_string
# Debug logging to verify the conversion
verbose_logger.debug(f"Converted budget_reset_at to ISO format: {iso_string}")
verbose_logger.debug("Converted budget_reset_at to ISO format: %s", iso_string)
except Exception as e:
verbose_logger.debug(f"Error processing budget reset datetime: {e}")
verbose_logger.debug(f"Original value: {user_api_key_budget_reset_at}")
verbose_logger.debug("Error processing budget reset datetime: %s", e)
verbose_logger.debug("Original value: %s", user_api_key_budget_reset_at)
return spend_metrics
@ -707,7 +707,7 @@ class DataDogLLMObsLogger(CustomBatchLogger):
kv_pairs[f"tool_calls.{idx}.function.arguments"] = json.dumps(function_arguments)
except (KeyError, TypeError, ValueError) as e:
verbose_logger.debug(f"DataDogLLMObs: Error processing tool call {idx}: {e!s}")
verbose_logger.debug("DataDogLLMObs: Error processing tool call %s: %s", idx, e)
continue
return kv_pairs
@ -747,6 +747,6 @@ class DataDogLLMObsLogger(CustomBatchLogger):
tool_call_metadata[f"output_{key}"] = value
except Exception as e:
verbose_logger.debug(f"DataDogLLMObs: Error extracting tool call metadata: {e!s}")
verbose_logger.debug("DataDogLLMObs: Error extracting tool call metadata: %s", e)
return tool_call_metadata

View file

@ -180,7 +180,7 @@ class DatadogMetricsLogger(CustomBatchLogger):
await self.flush_queue()
except Exception as e:
verbose_logger.exception(f"Datadog Metrics: Error in async_log_success_event: {e!s}")
verbose_logger.exception("Datadog Metrics: Error in async_log_success_event: %s", e)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
try:
@ -202,7 +202,7 @@ class DatadogMetricsLogger(CustomBatchLogger):
await self.flush_queue()
except Exception as e:
verbose_logger.exception(f"Datadog Metrics: Error in async_log_failure_event: {e!s}")
verbose_logger.exception("Datadog Metrics: Error in async_log_failure_event: %s", e)
async def async_send_batch(self):
if not self.log_queue:
@ -214,7 +214,7 @@ class DatadogMetricsLogger(CustomBatchLogger):
try:
await self._upload_to_datadog(payload_data)
except Exception as e:
verbose_logger.exception(f"Datadog Metrics: Error in async_send_batch: {e!s}")
verbose_logger.exception("Datadog Metrics: Error in async_send_batch: %s", e)
raise
async def _upload_to_datadog(self, payload: DatadogMetricsPayload):
@ -242,7 +242,7 @@ class DatadogMetricsLogger(CustomBatchLogger):
response.raise_for_status()
verbose_logger.debug(
f"Datadog Metrics: Uploaded {len(payload['series'])} metric points. Status: {response.status_code}"
"Datadog Metrics: Uploaded %s metric points. Status: %s", len(payload["series"]), response.status_code
)
async def async_health_check(self) -> IntegrationHealthCheckStatus:

View file

@ -23,9 +23,9 @@ def log_retry_error(details):
exception = details.get("exception")
tries = details.get("tries")
if exception:
logging.error(f"Confident AI Error: {exception}. Retrying: {tries} time(s)...")
logging.error("Confident AI Error: %s. Retrying: %s time(s)...", exception, tries)
else:
logging.error(f"Retrying: {tries} time(s)...")
logging.error("Retrying: %s time(s)...", tries)
class HttpMethods(Enum):

View file

@ -70,7 +70,7 @@ class DyanmoDBLogger:
# Assuming log_data is a dictionary with log information
response = table.put_item(Item=payload)
print_verbose(f"Response from DynamoDB:{response!s}")
print_verbose(f"Response from DynamoDB:{response}")
print_verbose(f"DynamoDB Layer Logging - final response object: {response_obj}")
return response

View file

@ -128,7 +128,7 @@ class GalileoObserve(CustomLogger):
except Exception as e:
return IntegrationHealthCheckStatus(
status="unhealthy",
error_message=f"Galileo health check failed: {e!s}",
error_message=f"Galileo health check failed: {e}",
)
async def async_set_galileo_headers(self) -> None:

View file

@ -76,7 +76,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
await self.log_queue.put(GCSLogQueueItem(payload=logging_payload, kwargs=kwargs, response_obj=response_obj))
except Exception as e:
verbose_logger.exception(f"GCS Bucket logging error: {e!s}")
verbose_logger.exception("GCS Bucket logging error: %s", e)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
try:
@ -95,7 +95,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
await self.log_queue.put(GCSLogQueueItem(payload=logging_payload, kwargs=kwargs, response_obj=response_obj))
except Exception as e:
verbose_logger.exception(f"GCS Bucket logging error: {e!s}")
verbose_logger.exception("GCS Bucket logging error: %s", e)
def _drain_queue_batch(self) -> list[GCSLogQueueItem]:
"""
@ -218,7 +218,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
except Exception as e:
success_count = 0
error_count = len(items)
verbose_logger.exception(f"GCS Bucket error logging batch payload to GCS bucket: {e!s}")
verbose_logger.exception("GCS Bucket error logging batch payload to GCS bucket: %s", e)
return (success_count, error_count)
async def _send_individual_logs(self, items: list[GCSLogQueueItem]) -> None:
@ -255,7 +255,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
logging_payload=item["payload"],
)
except Exception as e:
verbose_logger.exception(f"GCS Bucket error logging individual payload to GCS bucket: {e!s}")
verbose_logger.exception("GCS Bucket error logging individual payload to GCS bucket: %s", e)
async def async_send_batch(self):
"""
@ -336,7 +336,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
loaded_response = json.loads(response)
return loaded_response
except Exception as e:
verbose_logger.debug(f"Failed to fetch payload for date {date_str}: {e!s}")
verbose_logger.debug("Failed to fetch payload for date %s: %s", date_str, e)
continue
return None
@ -370,7 +370,7 @@ class GCSBucketLogger(GCSBucketBase, AdditionalLoggingUtils):
"""
while True:
await asyncio.sleep(self.flush_interval)
verbose_logger.debug(f"GCS Bucket periodic flush after {self.flush_interval} seconds")
verbose_logger.debug("GCS Bucket periodic flush after %s seconds", self.flush_interval)
await self.flush_queue()
async def async_health_check(self) -> IntegrationHealthCheckStatus:

View file

@ -45,7 +45,7 @@ async def _mock_async_handler_get(self, url, params=None, headers=None, follow_r
"""Monkey-patched AsyncHTTPHandler.get that intercepts GCS calls."""
# Only mock GCS API calls
if isinstance(url, str) and "storage.googleapis.com" in url:
verbose_logger.info(f"[GCS MOCK] GET to {url}")
verbose_logger.info("[GCS MOCK] GET to %s", url)
await asyncio.sleep(_MOCK_LATENCY_SECONDS)
# Return a minimal but valid StandardLoggingPayload JSON string as bytes
# This matches what GCS returns when downloading with ?alt=media
@ -117,7 +117,7 @@ async def _mock_async_handler_delete(
"""Monkey-patched AsyncHTTPHandler.delete that intercepts GCS calls."""
# Only mock GCS API calls
if isinstance(url, str) and "storage.googleapis.com" in url:
verbose_logger.info(f"[GCS MOCK] DELETE to {url}")
verbose_logger.info("[GCS MOCK] DELETE to %s", url)
await asyncio.sleep(_MOCK_LATENCY_SECONDS)
# DELETE returns 204 No Content with empty body (not JSON)
return MockResponse(

View file

@ -132,7 +132,7 @@ class GcsPubSubLogger(CustomBatchLogger):
await self.async_send_batch()
except Exception as e:
verbose_logger.exception(f"PubSub Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("PubSub Layer Error - %s\n%s", e, traceback.format_exc())
async def async_send_batch(self):
"""
@ -142,13 +142,13 @@ class GcsPubSubLogger(CustomBatchLogger):
if not self.log_queue:
return
verbose_logger.debug(f"PubSub - about to flush {len(self.log_queue)} events")
verbose_logger.debug("PubSub - about to flush %s events", len(self.log_queue))
for message in self.log_queue:
await self.publish_message(message)
except Exception as e:
verbose_logger.exception(f"PubSub Error sending batch - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("PubSub Error sending batch - %s\n%s", e, traceback.format_exc())
finally:
self.log_queue.clear()

View file

@ -42,7 +42,7 @@ def load_compatible_callbacks() -> dict:
with open(json_path, "r") as f:
return json.load(f)
except Exception as e:
verbose_logger.warning(f"Error loading generic_api_compatible_callbacks.json: {e!s}")
verbose_logger.warning("Error loading generic_api_compatible_callbacks.json: %s", e)
return {}
@ -124,7 +124,7 @@ class GenericAPILogger(CustomBatchLogger):
#########################################################
if callback_name:
if is_callback_compatible(callback_name):
verbose_logger.debug(f"Loading configuration for callback: {callback_name}")
verbose_logger.debug("Loading configuration for callback: %s", callback_name)
callback_config = get_callback_config(callback_name)
# Use config from JSON if not explicitly provided
@ -145,7 +145,7 @@ class GenericAPILogger(CustomBatchLogger):
log_format = callback_config["log_format"]
else:
verbose_logger.warning(
f"callback_name '{callback_name}' not found in generic_api_compatible_callbacks.json"
"callback_name '%s' not found in generic_api_compatible_callbacks.json", callback_name
)
#########################################################
@ -177,7 +177,12 @@ class GenericAPILogger(CustomBatchLogger):
self.log_format: LOG_FORMAT_TYPES = log_format or "json_array"
verbose_logger.debug(
f"in init GenericAPILogger, callback_name: {self.callback_name}, endpoint {self.endpoint}, headers {self.headers}, event_types: {self.event_types}, log_format: {self.log_format}"
"in init GenericAPILogger, callback_name: %s, endpoint %s, headers %s, event_types: %s, log_format: %s",
self.callback_name,
self.endpoint,
self.headers,
self.event_types,
self.log_format,
)
#########################################################
@ -214,7 +219,7 @@ class GenericAPILogger(CustomBatchLogger):
key, value = item.split("=", 1)
headers_dict[key.strip()] = value.strip()
except Exception as e:
verbose_logger.warning(f"Error parsing headers from environment variables: {e!s}")
verbose_logger.warning("Error parsing headers from environment variables: %s", e)
# 2. Update with litellm generic headers if available
if litellm.generic_logger_headers:
@ -308,7 +313,7 @@ class GenericAPILogger(CustomBatchLogger):
await self.async_send_batch()
except Exception as e:
verbose_logger.exception(f"Generic API Logger Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Generic API Logger Error - %s\n%s", e, traceback.format_exc())
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
"""
@ -339,7 +344,7 @@ class GenericAPILogger(CustomBatchLogger):
await self.async_send_batch()
except Exception as e:
verbose_logger.exception(f"Generic API Logger Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Generic API Logger Error - %s\n%s", e, traceback.format_exc())
async def async_send_batch(self):
"""
@ -355,7 +360,7 @@ class GenericAPILogger(CustomBatchLogger):
return
verbose_logger.debug(
f"Generic API Logger - about to flush {len(self.log_queue)} events in '{self.log_format}' format"
"Generic API Logger - about to flush %s events in '%s' format", len(self.log_queue), self.log_format
)
if self.log_format == "single":
@ -371,11 +376,13 @@ class GenericAPILogger(CustomBatchLogger):
# Log results
for idx, result in enumerate(responses):
if isinstance(result, Exception):
verbose_logger.exception(f"Generic API Logger - Error sending log {idx}: {result}")
verbose_logger.exception("Generic API Logger - Error sending log %s: %s", idx, result)
else:
# result is a Response object
verbose_logger.debug(
f"Generic API Logger - sent log {idx}, status: {result.status_code}" # type: ignore
"Generic API Logger - sent log %s, status: %s",
idx,
result.status_code, # type: ignore
)
else:
# Format the payload based on log_format
@ -390,12 +397,14 @@ class GenericAPILogger(CustomBatchLogger):
response = await self._post_with_retries(data=data)
verbose_logger.debug(
f"Generic API Logger - sent batch to {self.endpoint}, "
f"status: {response.status_code}, format: {self.log_format}"
"Generic API Logger - sent batch to %s, status: %s, format: %s",
self.endpoint,
response.status_code,
self.log_format,
)
except Exception as e:
verbose_logger.exception(f"Generic API Logger Error sending batch - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Generic API Logger Error sending batch - %s\n%s", e, traceback.format_exc())
finally:
self.log_queue.clear()
@ -405,7 +414,7 @@ class GenericAPILogger(CustomBatchLogger):
Returns a dict of the payload to send to the Generic API Endpoint
"""
verbose_logger.debug(f"GenericAPILogger Logging - Enters logging function for model {kwargs}")
verbose_logger.debug("GenericAPILogger Logging - Enters logging function for model %s", kwargs)
# construct payload to send custom logger
# follows the same params as langfuse.py

View file

@ -379,7 +379,7 @@ class GitLabPromptManager(CustomPromptManagement):
except Exception as e:
import litellm
litellm._logging.verbose_proxy_logger.error(f"Error in GitLab prompt pre_call_hook: {e}")
litellm._logging.verbose_proxy_logger.error("Error in GitLab prompt pre_call_hook: %s", e)
return messages, litellm_params
def _parse_prompt_to_messages(self, prompt_content: str) -> list[AllMessageValues]:

View file

@ -117,7 +117,7 @@ class LagoLogger(CustomLogger):
}
}
verbose_logger.debug(f"\033[91mLogged Lago Object:\n{returned_val}\033[0m\n")
verbose_logger.debug("\x1b[91mLogged Lago Object:\n%s\x1b[0m\n", returned_val)
return returned_val
def log_success_event(self, kwargs, response_obj, start_time, end_time):
@ -149,7 +149,7 @@ class LagoLogger(CustomLogger):
except Exception as e:
error_response = getattr(e, "response", None)
if error_response is not None and hasattr(error_response, "text"):
verbose_logger.debug(f"\nError Message: {error_response.text}")
verbose_logger.debug("\nError Message: %s", error_response.text)
raise e
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
@ -184,8 +184,8 @@ class LagoLogger(CustomLogger):
response.raise_for_status()
verbose_logger.debug(f"Logged Lago Object: {response.text}")
verbose_logger.debug("Logged Lago Object: %s", response.text)
except Exception as e:
if response is not None and hasattr(response, "text"):
verbose_logger.debug(f"\nError Message: {response.text}")
verbose_logger.debug("\nError Message: %s", response.text)
raise e

View file

@ -199,7 +199,7 @@ class LangFuseLogger:
)
langfuse_client = Langfuse(**parameters)
litellm.initialized_langfuse_clients += 1
verbose_logger.debug(f"Created langfuse client number {litellm.initialized_langfuse_clients}")
verbose_logger.debug("Created langfuse client number %s", litellm.initialized_langfuse_clients)
return langfuse_client
@staticmethod
@ -226,9 +226,9 @@ class LangFuseLogger:
if metadata_param_key.startswith("langfuse_"):
trace_param_key = metadata_param_key.replace("langfuse_", "", 1)
if trace_param_key in metadata:
verbose_logger.warning(f"Overwriting Langfuse `{trace_param_key}` from request header")
verbose_logger.warning("Overwriting Langfuse `%s` from request header", trace_param_key)
else:
verbose_logger.debug(f"Found Langfuse `{trace_param_key}` in request header")
verbose_logger.debug("Found Langfuse `%s` in request header", trace_param_key)
metadata[trace_param_key] = proxy_headers.get(metadata_param_key)
return metadata
@ -256,7 +256,7 @@ class LangFuseLogger:
Logs a success or error event on Langfuse
"""
try:
verbose_logger.debug(f"Langfuse Logging - Enters logging function for model {kwargs}")
verbose_logger.debug("Langfuse Logging - Enters logging function for model %s", kwargs)
# set default values for input/output for langfuse logging
input = None
@ -295,7 +295,7 @@ class LangFuseLogger:
level=level,
status_message=status_message,
)
verbose_logger.debug(f"OUTPUT IN LANGFUSE: {output}; original: {response_obj}")
verbose_logger.debug("OUTPUT IN LANGFUSE: %s; original: %s", output, response_obj)
trace_id = None
generation_id = None
if self._is_langfuse_v2():
@ -325,12 +325,12 @@ class LangFuseLogger:
input=input,
response_obj=response_obj,
)
verbose_logger.debug(f"Langfuse Layer Logging - final response object: {response_obj}")
verbose_logger.debug("Langfuse Layer Logging - final response object: %s", response_obj)
verbose_logger.info("Langfuse Layer Logging - logging success")
return {"trace_id": trace_id, "generation_id": generation_id}
except Exception as e:
verbose_logger.exception(f"Langfuse Layer Error(): Exception occured - {e!s}")
verbose_logger.exception("Langfuse Layer Error(): Exception occured - %s", e)
return {"trace_id": None, "generation_id": None}
def _get_langfuse_input_output_content(
@ -625,7 +625,7 @@ class LangFuseLogger:
trace_params["metadata"] = {"metadata_passed_to_litellm": metadata}
cost = kwargs.get("response_cost", None)
verbose_logger.debug(f"trace: {cost}")
verbose_logger.debug("trace: %s", cost)
clean_metadata["litellm_response_cost"] = cost
if standard_logging_object is not None:
@ -780,12 +780,13 @@ class LangFuseLogger:
if hasattr(generation_client, "trace_id") and generation_client.trace_id:
if generation_client.trace_id != trace_id:
verbose_logger.warning(
f"Langfuse trace_id mismatch: set {trace_id}, but langfuse returned {generation_client.trace_id}. "
"Using our intended trace_id for consistency."
"Langfuse trace_id mismatch: set %s, but langfuse returned %s. Using our intended trace_id for consistency.",
trace_id,
generation_client.trace_id,
)
return trace_id, generation_id
except Exception:
verbose_logger.error(f"Langfuse Layer Error - {traceback.format_exc()}")
verbose_logger.error("Langfuse Layer Error - %s", traceback.format_exc())
return None, None
@staticmethod
@ -902,7 +903,7 @@ class LangFuseLogger:
# For other types, try to apply the function directly
return masking_function(data)
except Exception as e:
verbose_logger.warning(f"Failed to apply masking function: {e}. Returning original data.")
verbose_logger.warning("Failed to apply masking function: %s. Returning original data.", e)
return data
@staticmethod
@ -966,7 +967,7 @@ class LangFuseLogger:
end_time=guardrail_entry.get("end_time", None), # type: ignore
)
verbose_logger.debug(f"Logged guardrail information as span: {span}")
verbose_logger.debug("Logged guardrail information as span: %s", span)
span.end()
@ -1035,7 +1036,7 @@ def _add_prompt_to_generation_params(
try:
generation_params["prompt"] = langfuse_client.get_prompt(prompt_management_metadata["prompt_id"])
except Exception as e:
verbose_logger.debug(f"[Non-blocking] Langfuse Logger: Error getting prompt client for logging: {e}")
verbose_logger.debug("[Non-blocking] Langfuse Logger: Error getting prompt client for logging: %s", e)
else:
generation_params["prompt"] = user_prompt

View file

@ -315,10 +315,10 @@ class LangfuseOtelLogger(OpenTelemetry):
if langfuse_host:
normalized_host = langfuse_host if langfuse_host.startswith("http") else f"https://{langfuse_host}"
endpoint = f"{normalized_host.rstrip('/')}/api/public/otel"
verbose_logger.debug(f"Using Langfuse OTEL endpoint from host: {endpoint}")
verbose_logger.debug("Using Langfuse OTEL endpoint from host: %s", endpoint)
else:
endpoint = LANGFUSE_CLOUD_US_ENDPOINT
verbose_logger.debug(f"Using Langfuse US cloud endpoint: {endpoint}")
verbose_logger.debug("Using Langfuse US cloud endpoint: %s", endpoint)
auth_header = LangfuseOtelLogger._get_langfuse_authorization_header(
public_key=public_key, secret_key=secret_key

View file

@ -317,7 +317,7 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
except Exception as e:
from litellm._logging import verbose_logger
verbose_logger.exception(f"Langfuse Layer Error - Exception occurred while logging success event: {e!s}")
verbose_logger.exception("Langfuse Layer Error - Exception occurred while logging success event: %s", e)
self.handle_callback_failure(callback_name="langfuse")
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
@ -347,5 +347,5 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
except Exception as e:
from litellm._logging import verbose_logger
verbose_logger.exception(f"Langfuse Layer Error - Exception occurred while logging failure event: {e!s}")
verbose_logger.exception("Langfuse Layer Error - Exception occurred while logging failure event: %s", e)
self.handle_callback_failure(callback_name="langfuse")

View file

@ -194,7 +194,7 @@ class LangsmithLogger(CustomBatchLogger):
fields = self._extract_metadata_fields(metadata, credentials)
verbose_logger.debug(
f"Langsmith Logging - project_name: {fields['project_name']}, run_name {fields['run_name']}"
"Langsmith Logging - project_name: %s, run_name %s", fields["project_name"], fields["run_name"]
)
payload: StandardLoggingPayload | None = kwargs.get("standard_logging_object", None)
@ -244,7 +244,7 @@ class LangsmithLogger(CustomBatchLogger):
random_sample = random.random()
if random_sample > sampling_rate:
verbose_logger.info(
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
)
return # Skip logging
verbose_logger.debug(
@ -267,7 +267,7 @@ class LangsmithLogger(CustomBatchLogger):
credentials=credentials,
)
)
verbose_logger.debug(f"Langsmith, event added to queue. Will flush in {self.flush_interval} seconds...")
verbose_logger.debug("Langsmith, event added to queue. Will flush in %s seconds...", self.flush_interval)
if len(self.log_queue) >= self.batch_size:
self._send_batch()
@ -282,7 +282,7 @@ class LangsmithLogger(CustomBatchLogger):
random_sample = random.random()
if random_sample > sampling_rate:
verbose_logger.info(
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
)
return # Skip logging
verbose_logger.debug(
@ -321,7 +321,7 @@ class LangsmithLogger(CustomBatchLogger):
random_sample = random.random()
if random_sample > sampling_rate:
verbose_logger.info(
f"Skipping Langsmith logging. Sampling rate={sampling_rate}, random_sample={random_sample}"
"Skipping Langsmith logging. Sampling rate=%s, random_sample=%s", sampling_rate, random_sample
)
return # Skip logging
verbose_logger.info("Langsmith Failure Event Logging!")
@ -422,16 +422,16 @@ class LangsmithLogger(CustomBatchLogger):
response.raise_for_status()
if response.status_code >= 300:
verbose_logger.error(f"Langsmith Error: {response.status_code} - {response.text}")
verbose_logger.error("Langsmith Error: %s - %s", response.status_code, response.text)
else:
if self.is_mock_mode:
verbose_logger.debug(f"[LANGSMITH MOCK] Batch of {len(elements_to_log)} runs successfully mocked")
verbose_logger.debug("[LANGSMITH MOCK] Batch of %s runs successfully mocked", len(elements_to_log))
else:
verbose_logger.debug(f"Batch of {len(self.log_queue)} runs successfully created")
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
except httpx.HTTPStatusError as e:
verbose_logger.exception(f"Langsmith HTTP Error: {e.response.status_code} - {e.response.text}")
verbose_logger.exception("Langsmith HTTP Error: %s - %s", e.response.status_code, e.response.text)
except Exception:
verbose_logger.exception(f"Langsmith Layer Error - {traceback.format_exc()}")
verbose_logger.exception("Langsmith Layer Error - %s", traceback.format_exc())
def _group_batches_by_credentials(self) -> dict[CredentialsKey, BatchGroup]:
"""Groups queue objects by credentials using a proper key structure"""

View file

@ -94,9 +94,9 @@ class LiteralAILogger(CustomBatchLogger):
)
if response.status_code >= 300:
verbose_logger.error(f"Literal AI Error: {response.status_code} - {response.text}")
verbose_logger.error("Literal AI Error: %s - %s", response.status_code, response.text)
else:
verbose_logger.debug(f"Batch of {len(self.log_queue)} runs successfully created")
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
except Exception:
verbose_logger.exception("Literal AI Layer Error")
@ -152,11 +152,11 @@ class LiteralAILogger(CustomBatchLogger):
headers=self.headers,
)
if response.status_code >= 300:
verbose_logger.error(f"Literal AI Error: {response.status_code} - {response.text}")
verbose_logger.error("Literal AI Error: %s - %s", response.status_code, response.text)
else:
verbose_logger.debug(f"Batch of {len(self.log_queue)} runs successfully created")
verbose_logger.debug("Batch of %s runs successfully created", len(self.log_queue))
except httpx.HTTPStatusError as e:
verbose_logger.exception(f"Literal AI HTTP Error: {e.response.status_code} - {e.response.text}")
verbose_logger.exception("Literal AI HTTP Error: %s - %s", e.response.status_code, e.response.text)
except Exception:
verbose_logger.exception("Literal AI Layer Error")

View file

@ -35,7 +35,7 @@ class LogfireLogger:
if logfire.DEFAULT_LOGFIRE_INSTANCE.config.send_to_logfire:
logfire.configure(token=os.getenv("LOGFIRE_TOKEN"))
except Exception as e:
print_verbose(f"Got exception on init logfire client {e!s}")
print_verbose(f"Got exception on init logfire client {e}")
raise e
def _get_span_config(self, payload) -> SpanConfig:
@ -90,7 +90,7 @@ class LogfireLogger:
try:
import logfire
verbose_logger.debug(f"logfire Logging - Enters logging function for model {kwargs}")
verbose_logger.debug("logfire Logging - Enters logging function for model %s", kwargs)
if not response_obj:
response_obj = {}
@ -159,4 +159,4 @@ class LogfireLogger:
print_verbose(f"Logfire Layer Logging - final response object: {response_obj}")
except Exception as e:
verbose_logger.debug(f"Logfire Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.debug("Logfire Layer Error - %s\n%s", e, traceback.format_exc())

View file

@ -99,7 +99,7 @@ class MlflowLogger(CustomLogger):
)
except Exception as e:
verbose_logger.debug(f"MLflow Logging Error - {e}", stack_info=True)
verbose_logger.debug("MLflow Logging Error - %s", e, stack_info=True)
def _handle_stream_event(self, kwargs, response_obj, start_time, end_time):
"""

View file

@ -144,7 +144,7 @@ def create_mock_client_factory(config: MockClientConfig):
):
"""Monkey-patched AsyncHTTPHandler.post that intercepts API calls."""
if isinstance(url, str) and _is_mock_url(url):
verbose_logger.info(f"[{config.name} MOCK] POST to {url}")
verbose_logger.info("[%s MOCK] POST to %s", config.name, url)
await asyncio.sleep(_MOCK_LATENCY_SECONDS)
return MockResponse(
status_code=config.default_status_code,
@ -172,7 +172,7 @@ def create_mock_client_factory(config: MockClientConfig):
def _mock_sync_client_post(self, url, **kwargs):
"""Monkey-patched httpx.Client.post that intercepts API calls."""
if _is_mock_url(url):
verbose_logger.info(f"[{config.name} MOCK] POST to {url} (sync)")
verbose_logger.info("[%s MOCK] POST to %s (sync)", config.name, url)
return MockResponse(
status_code=config.default_status_code,
json_data=config.default_json_data,
@ -198,7 +198,7 @@ def create_mock_client_factory(config: MockClientConfig):
):
"""Monkey-patched HTTPHandler.post that intercepts API calls."""
if isinstance(url, str) and _is_mock_url(url):
verbose_logger.info(f"[{config.name} MOCK] POST to {url}")
verbose_logger.info("[%s MOCK] POST to %s", config.name, url)
import time
time.sleep(_MOCK_LATENCY_SECONDS)
@ -236,29 +236,29 @@ def create_mock_client_factory(config: MockClientConfig):
if _mocks_initialized:
return
verbose_logger.debug(f"[{config.name} MOCK] Initializing {config.name} mock client...")
verbose_logger.debug("[%s MOCK] Initializing %s mock client...", config.name, config.name)
if config.patch_async_handler and _original_async_handler_post is None:
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
_original_async_handler_post = AsyncHTTPHandler.post
AsyncHTTPHandler.post = _mock_async_handler_post # type: ignore
verbose_logger.debug(f"[{config.name} MOCK] Patched AsyncHTTPHandler.post")
verbose_logger.debug("[%s MOCK] Patched AsyncHTTPHandler.post", config.name)
if config.patch_sync_client and _original_sync_client_post is None:
_original_sync_client_post = httpx.Client.post
httpx.Client.post = _mock_sync_client_post # type: ignore
verbose_logger.debug(f"[{config.name} MOCK] Patched httpx.Client.post")
verbose_logger.debug("[%s MOCK] Patched httpx.Client.post", config.name)
if config.patch_http_handler and _original_http_handler_post is None:
from litellm.llms.custom_httpx.http_handler import HTTPHandler
_original_http_handler_post = HTTPHandler.post
HTTPHandler.post = _mock_http_handler_post # type: ignore
verbose_logger.debug(f"[{config.name} MOCK] Patched HTTPHandler.post")
verbose_logger.debug("[%s MOCK] Patched HTTPHandler.post", config.name)
verbose_logger.debug(f"[{config.name} MOCK] Mock latency set to {_MOCK_LATENCY_SECONDS * 1000:.0f}ms")
verbose_logger.debug(f"[{config.name} MOCK] {config.name} mock client initialization complete")
verbose_logger.debug("[%s MOCK] %s mock client initialization complete", config.name, config.name)
_mocks_initialized = True
@ -274,7 +274,7 @@ def create_mock_client_factory(config: MockClientConfig):
result = bool(result) if result is not None else False
if result:
verbose_logger.info(f"{config.name} Mock Mode: ENABLED - API calls will be mocked")
verbose_logger.info("%s Mock Mode: ENABLED - API calls will be mocked", config.name)
return result

View file

@ -116,11 +116,12 @@ class NewRelicLogger(CustomLogger):
self.enabled = True
verbose_logger.info(
f"New Relic AI Monitoring initialized for app: {self.app_name}, "
f"content recording: {self.record_content}"
"New Relic AI Monitoring initialized for app: %s, content recording: %s",
self.app_name,
self.record_content,
)
except Exception as e:
verbose_logger.error(f"Failed to initialize New Relic agent: {e}. Integration will be disabled.")
verbose_logger.error("Failed to initialize New Relic agent: %s. Integration will be disabled.", e)
self.enabled = False
def _get_newrelic_params(self) -> dict:
@ -170,9 +171,10 @@ class NewRelicLogger(CustomLogger):
if value in ("0", "false", "no", "off"):
return False
verbose_logger.warning(
f"{var_name}={raw!r} is not a recognised boolean "
f"(accepts true/false, 1/0, yes/no, on/off). "
f"Falling back to default ({default})."
"%s=%r is not a recognised boolean (accepts true/false, 1/0, yes/no, on/off). Falling back to default (%s).",
var_name,
raw,
default,
)
return default
@ -188,7 +190,7 @@ class NewRelicLogger(CustomLogger):
return version("litellm")
except Exception as e:
verbose_logger.warning(f"Unable to determine litellm version: {e}")
verbose_logger.warning("Unable to determine litellm version: %s", e)
return "unknown"
def _emit_supportability_metric(self):
@ -216,12 +218,12 @@ class NewRelicLogger(CustomLogger):
if app and app.enabled:
app.record_custom_metric(metric_name, 1)
verbose_logger.info(f"Emitted New Relic supportability metric: {metric_name}")
verbose_logger.info("Emitted New Relic supportability metric: %s", metric_name)
else:
verbose_logger.info("New Relic application is not enabled; skipping metric recording.")
except Exception as e:
verbose_logger.warning(f"Failed to emit supportability metric: {e}")
verbose_logger.warning("Failed to emit supportability metric: %s", e)
def _check_and_emit_periodic_metric(self):
"""
@ -294,14 +296,13 @@ class NewRelicLogger(CustomLogger):
trace_id = slo_trace_id
except Exception as e:
verbose_logger.warning(f"Unable to parse New Relic trace context from upstream sources: {e}")
verbose_logger.warning("Unable to parse New Relic trace context from upstream sources: %s", e)
if not trace_id:
trace_id = uuid.uuid4().hex
verbose_logger.debug(
f"New Relic trace_id not available from distributed tracing headers or "
f"StandardLoggingPayload. Generated trace_id={trace_id} for AI monitoring "
f"event grouping."
"New Relic trace_id not available from distributed tracing headers or StandardLoggingPayload. Generated trace_id=%s for AI monitoring event grouping.",
trace_id,
)
return trace_id
@ -638,7 +639,7 @@ class NewRelicLogger(CustomLogger):
verbose_logger.warning("New Relic application is not enabled; skipping summary event recording.")
except Exception as e:
verbose_logger.warning(f"Failed to record New Relic summary event: {e}")
verbose_logger.warning("Failed to record New Relic summary event: %s", e)
self.handle_callback_failure("newrelic")
def _record_message_events(
@ -699,7 +700,7 @@ class NewRelicLogger(CustomLogger):
app.record_custom_event("LlmChatCompletionMessage", event_data)
except Exception as e:
verbose_logger.warning(f"Failed to record New Relic message events: {e}")
verbose_logger.warning("Failed to record New Relic message events: %s", e)
self.handle_callback_failure("newrelic")
def _record_error_metric(self):
@ -714,7 +715,7 @@ class NewRelicLogger(CustomLogger):
if app and app.enabled:
app.record_custom_metric("LLM/LiteLLM/Error", 1)
except Exception as e:
verbose_logger.warning(f"Failed to record New Relic error metric: {e}")
verbose_logger.warning("Failed to record New Relic error metric: %s", e)
self.handle_callback_failure("newrelic")
def _process_success(
@ -846,7 +847,7 @@ class NewRelicLogger(CustomLogger):
try:
self._process_success(kwargs, response_obj, start_time, end_time)
except Exception as e:
verbose_logger.warning(f"Error in New Relic log_success_event: {e}")
verbose_logger.warning("Error in New Relic log_success_event: %s", e)
self.handle_callback_failure("newrelic")
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
@ -859,7 +860,7 @@ class NewRelicLogger(CustomLogger):
try:
self._process_success(kwargs, response_obj, start_time, end_time)
except Exception as e:
verbose_logger.warning(f"Error in New Relic async_log_success_event: {e}")
verbose_logger.warning("Error in New Relic async_log_success_event: %s", e)
self.handle_callback_failure("newrelic")
def log_failure_event(self, kwargs, response_obj, start_time, end_time):
@ -872,7 +873,7 @@ class NewRelicLogger(CustomLogger):
self._record_error_metric()
except Exception as e:
verbose_logger.warning(f"Error in New Relic log_failure_event: {e}")
verbose_logger.warning("Error in New Relic log_failure_event: %s", e)
self.handle_callback_failure("newrelic")
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
@ -885,5 +886,5 @@ class NewRelicLogger(CustomLogger):
self._record_error_metric()
except Exception as e:
verbose_logger.warning(f"Error in New Relic async_log_failure_event: {e}")
verbose_logger.warning("Error in New Relic async_log_failure_event: %s", e)
self.handle_callback_failure("newrelic")

View file

@ -24,6 +24,10 @@ from litellm.integrations.opentelemetry_utils.gen_ai_semconv import (
from litellm.integrations.otel.model.semconv import Metric
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.litellm_core_utils.secret_redaction import redact_string
from litellm.litellm_core_utils.service_tier_utils import (
get_requested_service_tier,
get_served_service_tier,
)
from litellm.secret_managers.main import get_secret_bool, str_to_bool
from litellm.types.services import ServiceLoggerPayload
from litellm.types.utils import (
@ -74,6 +78,11 @@ PREPROCESSING_DURATION_MS_ATTRIBUTE = "litellm.preprocessing.duration_ms"
TEAM_METADATA_ATTRIBUTE = "litellm.team.metadata"
MODEL_GROUP_ATTRIBUTE = "litellm.model_group"
PROVIDER_MODEL_ATTRIBUTE = "litellm.provider.model"
# semconv names the service tier attributes under the openai namespace, but every
# provider that reports a tier (OpenAI, Anthropic, Bedrock, Groq, Vertex) uses the
# same request param and response field, so both keys carry all of them.
REQUEST_SERVICE_TIER_ATTRIBUTE = "gen_ai.openai.request.service_tier"
RESPONSE_SERVICE_TIER_ATTRIBUTE = "gen_ai.openai.response.service_tier"
# Remove the hardcoded LITELLM_RESOURCE dictionary - we'll create it properly later
RAW_REQUEST_SPAN_NAME = "raw_gen_ai_request"
LITELLM_REQUEST_SPAN_NAME = "litellm_request"
@ -1411,6 +1420,23 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger):
if provider_model:
self.safe_set_attribute(span=span, key=PROVIDER_MODEL_ATTRIBUTE, value=provider_model)
def _set_service_tier_attributes(
self,
span: Span,
standard_logging_payload: StandardLoggingPayload,
) -> None:
"""Stamp the tier the caller asked for and the tier the provider reports it
served, so tier usage is segmentable in traces. Both are optional: a caller
may not name a tier, and streaming responses carry no served tier.
"""
requested_tier = get_requested_service_tier(standard_logging_payload)
if requested_tier is not None:
self.safe_set_attribute(span=span, key=REQUEST_SERVICE_TIER_ATTRIBUTE, value=requested_tier)
served_tier = get_served_service_tier(standard_logging_payload)
if served_tier is not None:
self.safe_set_attribute(span=span, key=RESPONSE_SERVICE_TIER_ATTRIBUTE, value=served_tier)
@staticmethod
def _team_metadata_json(value: Any, allowed_keys: list[str]) -> str | None:
"""JSON-serialize only the allowlisted sub-keys of a team's metadata.
@ -2310,6 +2336,8 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger):
value=response_obj.get("model"),
)
self._set_service_tier_attributes(span=span, standard_logging_payload=standard_logging_payload)
usage = response_obj and response_obj.get("usage")
if usage:
self.safe_set_attribute(
@ -2688,7 +2716,8 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger):
)
except json.JSONDecodeError:
verbose_logger.debug(
f"litellm.integrations.opentelemetry.py::set_raw_request_attributes() - raw_response not json string - {_raw_response}"
"litellm.integrations.opentelemetry.py::set_raw_request_attributes() - raw_response not json string - %s",
_raw_response,
)
self.safe_set_attribute(

View file

@ -81,7 +81,7 @@ class OpikLogger(CustomBatchLogger):
self.flush_lock: asyncio.Lock | None = asyncio.Lock()
except Exception as e:
verbose_logger.exception(
f"OpikLogger - Asynchronous processing not initialized as we are not running in an async context {e!s}"
"OpikLogger - Asynchronous processing not initialized as we are not running in an async context %s", e
)
self.flush_lock = None
@ -154,14 +154,14 @@ class OpikLogger(CustomBatchLogger):
self.log_queue.append(span_payload.__dict__)
verbose_logger.debug(
f"OpikLogger added event to log_queue - Will flush in {self.flush_interval} seconds..."
"OpikLogger added event to log_queue - Will flush in %s seconds...", self.flush_interval
)
if len(self.log_queue) >= self.batch_size:
verbose_logger.debug("OpikLogger - Flushing batch")
await self.flush_queue()
except Exception as e:
verbose_logger.exception(f"OpikLogger failed to log success event - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("OpikLogger failed to log success event - %s\n%s", e, traceback.format_exc())
def _sync_send(self, url: str, headers: dict[str, str], batch: dict[str, Any]) -> None:
try:
@ -174,7 +174,7 @@ class OpikLogger(CustomBatchLogger):
if response.status_code != 204:
raise Exception(f"Response from opik API status_code: {response.status_code}, text: {response.text}")
except Exception as e:
verbose_logger.exception(f"OpikLogger failed to send batch - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("OpikLogger failed to send batch - %s\n%s", e, traceback.format_exc())
def log_success_event(
self,
@ -245,7 +245,7 @@ class OpikLogger(CustomBatchLogger):
batch={"spans": [span_payload.__dict__]},
)
except Exception as e:
verbose_logger.exception(f"OpikLogger failed to log success event - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("OpikLogger failed to log success event - %s\n%s", e, traceback.format_exc())
async def _submit_batch(self, url: str, headers: dict[str, str], batch: dict[str, Any]) -> None:
try:
@ -257,11 +257,11 @@ class OpikLogger(CustomBatchLogger):
response.raise_for_status()
if response.status_code >= 300:
verbose_logger.error(f"OpikLogger - Error: {response.status_code} - {response.text}")
verbose_logger.error("OpikLogger - Error: %s - %s", response.status_code, response.text)
else:
verbose_logger.info(f"OpikLogger - {len(self.log_queue)} Opik events submitted")
verbose_logger.info("OpikLogger - %s Opik events submitted", len(self.log_queue))
except Exception as e:
verbose_logger.exception(f"OpikLogger failed to send batch - {e!s}")
verbose_logger.exception("OpikLogger failed to send batch - %s", e)
def _create_opik_headers(self) -> dict[str, str]:
headers: dict[str, str] = {}
@ -283,7 +283,7 @@ class OpikLogger(CustomBatchLogger):
# Send trace batch
if len(traces) > 0:
await self._submit_batch(url=self.trace_url, headers=self.headers, batch={"traces": traces})
verbose_logger.info(f"Sent {len(traces)} traces")
verbose_logger.info("Sent %s traces", len(traces))
if len(spans) > 0:
await self._submit_batch(url=self.span_url, headers=self.headers, batch={"spans": spans})
verbose_logger.info(f"Sent {len(spans)} spans")
verbose_logger.info("Sent %s spans", len(spans))

View file

@ -66,7 +66,7 @@ def extract_opik_metadata(
if requester_opik:
opik_meta.update(requester_opik)
_logging.verbose_logger.debug(f"litellm_opik_metadata - {json.dumps(opik_meta, default=str)}")
_logging.verbose_logger.debug("litellm_opik_metadata - %s", json.dumps(opik_meta, default=str))
return opik_meta
@ -92,7 +92,7 @@ def extract_span_identifiers(
try:
return current_span_data.trace_id, current_span_data.id
except AttributeError:
_logging.verbose_logger.warning(f"Unexpected current_span_data format: {type(current_span_data)}")
_logging.verbose_logger.warning("Unexpected current_span_data format: %s", type(current_span_data))
return None, None
@ -152,7 +152,7 @@ def apply_proxy_header_overrides(
if isinstance(parsed_tags, list):
tags.extend(parsed_tags)
except (json.JSONDecodeError, TypeError):
_logging.verbose_logger.warning(f"Failed to parse tags from header: {value}")
_logging.verbose_logger.warning("Failed to parse tags from header: %s", value)
return project_name, tags, thread_id

View file

@ -61,7 +61,7 @@ def build_span_payload(
created = response_obj.get("created", 0)
span_name = f"{model}_{obj_type}_{created}"
_logging.verbose_logger.debug(f"OpikLogger creating span with id {span_id} for trace {trace_id}")
_logging.verbose_logger.debug("OpikLogger creating span with id %s for trace %s", span_id, trace_id)
return types.SpanPayload(
id=span_id,

View file

@ -72,7 +72,7 @@ class PostHogLogger(CustomBatchLogger):
super().__init__(**kwargs, flush_lock=None, batch_size=POSTHOG_MAX_BATCH_SIZE)
except Exception as e:
verbose_logger.exception(f"PostHog: Got exception on init PostHog client {e!s}")
verbose_logger.exception("PostHog: Got exception on init PostHog client %s", e)
raise e
def log_success_event(self, kwargs, response_obj, start_time, end_time):
@ -107,7 +107,7 @@ class PostHogLogger(CustomBatchLogger):
verbose_logger.debug("PostHog: Sync event successfully sent")
except Exception as e:
verbose_logger.exception(f"PostHog Sync Layer Error - {e!s}")
verbose_logger.exception("PostHog Sync Layer Error - %s", e)
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
try:
@ -115,7 +115,7 @@ class PostHogLogger(CustomBatchLogger):
self._ensure_async_setup() # Lazy initialization
await self._log_async_event(kwargs, response_obj, start_time, end_time)
except Exception as e:
verbose_logger.exception(f"PostHog Layer Error - {e!s}")
verbose_logger.exception("PostHog Layer Error - %s", e)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
try:
@ -123,7 +123,7 @@ class PostHogLogger(CustomBatchLogger):
self._ensure_async_setup() # Lazy initialization
await self._log_async_event(kwargs, response_obj, start_time, end_time)
except Exception as e:
verbose_logger.exception(f"PostHog Layer Error - {e!s}")
verbose_logger.exception("PostHog Layer Error - %s", e)
async def _log_async_event(self, kwargs, response_obj=None, start_time=0.0, end_time=0.0):
# Note: response_obj, start_time, end_time not used - all data comes from kwargs
@ -132,7 +132,7 @@ class PostHogLogger(CustomBatchLogger):
# Store event with its credentials for batch sending
self.log_queue.append({"event": event_payload, "api_key": api_key, "api_url": api_url})
verbose_logger.debug(f"PostHog, event added to queue. Will flush in {self.flush_interval} seconds...")
verbose_logger.debug("PostHog, event added to queue. Will flush in %s seconds...", self.flush_interval)
if len(self.log_queue) >= self.batch_size:
await self.flush_queue()
@ -328,7 +328,7 @@ class PostHogLogger(CustomBatchLogger):
if not self.log_queue:
return
verbose_logger.debug(f"PostHog: Sending batch of {len(self.log_queue)} events")
verbose_logger.debug("PostHog: Sending batch of %s events", len(self.log_queue))
if self.is_mock_mode:
verbose_logger.debug("[POSTHOG MOCK] Mock mode enabled - API calls will be intercepted")
@ -363,11 +363,11 @@ class PostHogLogger(CustomBatchLogger):
)
if self.is_mock_mode:
verbose_logger.debug(f"[POSTHOG MOCK] Batch of {len(self.log_queue)} events successfully mocked")
verbose_logger.debug("[POSTHOG MOCK] Batch of %s events successfully mocked", len(self.log_queue))
else:
verbose_logger.debug(f"PostHog: Batch of {len(self.log_queue)} events successfully sent")
verbose_logger.debug("PostHog: Batch of %s events successfully sent", len(self.log_queue))
except Exception as e:
verbose_logger.exception(f"PostHog Error sending batch API - {e!s}")
verbose_logger.exception("PostHog Error sending batch API - %s", e)
def _ensure_async_setup(self):
if not self._async_initialized:
@ -377,7 +377,7 @@ class PostHogLogger(CustomBatchLogger):
self._async_initialized = True
verbose_logger.debug("PostHog: Async components initialized")
except Exception as e:
verbose_logger.error(f"PostHog: Failed to initialize async components: {e!s}")
verbose_logger.error("PostHog: Failed to initialize async components: %s", e)
raise
def _extract_metadata(self, kwargs: dict[str, Any]) -> dict[str, Any]:
@ -408,7 +408,7 @@ class PostHogLogger(CustomBatchLogger):
if not self.log_queue:
return
verbose_logger.debug(f"PostHog: Flushing {len(self.log_queue)} remaining events on exit")
verbose_logger.debug("PostHog: Flushing %s remaining events on exit", len(self.log_queue))
try:
# Group events by credentials (same logic as async_send_batch)
@ -436,13 +436,13 @@ class PostHogLogger(CustomBatchLogger):
response.raise_for_status()
if response.status_code != 200:
verbose_logger.error(f"PostHog: Failed to flush on exit - status {response.status_code}")
verbose_logger.error("PostHog: Failed to flush on exit - status %s", response.status_code)
if self.is_mock_mode:
verbose_logger.debug(f"[POSTHOG MOCK] Successfully flushed {len(self.log_queue)} events on exit")
verbose_logger.debug("[POSTHOG MOCK] Successfully flushed %s events on exit", len(self.log_queue))
else:
verbose_logger.debug(f"PostHog: Successfully flushed {len(self.log_queue)} events on exit")
verbose_logger.debug("PostHog: Successfully flushed %s events on exit", len(self.log_queue))
self.log_queue.clear()
except Exception as e:
verbose_logger.error(f"PostHog: Error flushing events on exit: {e!s}")
verbose_logger.error("PostHog: Error flushing events on exit: %s", e)

View file

@ -34,6 +34,9 @@ from litellm.litellm_core_utils.core_helpers import (
get_litellm_metadata_from_kwargs,
get_metadata_variable_name_from_kwargs,
)
from litellm.litellm_core_utils.service_tier_utils import (
get_service_tier_from_standard_logging_payload,
)
from litellm.proxy._types import (
LiteLLM_DeletedVerificationToken,
LiteLLM_TeamTable,
@ -98,16 +101,6 @@ class _ExcludedLabelMetric:
return self._metric.labels(*kept_values) if kept_values else self._metric
# Tiers a caller may name in a request, across the providers that accept the
# parameter: OpenAI ("auto", "default", "flex", "priority", "scale"), Bedrock and
# Groq (subsets of those), Anthropic ("auto", "standard_only") and Vertex, which
# maps "default" to "standard". Used to bound the caller-controlled fallback in
# ``get_service_tier_from_standard_logging_payload``.
KNOWN_REQUEST_SERVICE_TIERS = frozenset(
{"auto", "batch", "default", "flex", "priority", "scale", "standard", "standard_only"}
)
def _get_budget_metrics_per_request_timeout() -> float:
raw = os.getenv("PROMETHEUS_BUDGET_METRICS_PER_REQUEST_TIMEOUT")
if raw is None:
@ -683,7 +676,7 @@ class PrometheusLogger(CustomLogger):
)
except Exception as e:
print_verbose(f"Got exception on init prometheus client {e!s}")
print_verbose(f"Got exception on init prometheus client {e}")
raise e
def _parse_prometheus_config(self) -> dict[str, list[str]]:
@ -697,7 +690,7 @@ class PrometheusLogger(CustomLogger):
if not config:
return {}
verbose_logger.debug(f"prometheus config: {config}")
verbose_logger.debug("prometheus config: %s", config)
# Parse and validate all configuration groups
parsed_configs = []
@ -963,7 +956,10 @@ class PrometheusLogger(CustomLogger):
except ImportError:
# Fallback to simple logging if rich is not available
verbose_logger.error(
f"Invalid labels for metric '{metric_name}': {invalid_labels}. Valid labels: {sorted(valid_labels)}"
"Invalid labels for metric '%s': %s. Valid labels: %s",
metric_name,
invalid_labels,
sorted(valid_labels),
)
def _pretty_print_invalid_metric_error(self, invalid_metric_name: str, valid_metrics: tuple) -> None:
@ -1003,7 +999,9 @@ class PrometheusLogger(CustomLogger):
except ImportError:
# Fallback to simple logging if rich is not available
verbose_logger.error(f"Invalid metric name: {invalid_metric_name}. Valid metrics: {sorted(valid_metrics)}")
verbose_logger.error(
"Invalid metric name: %s. Valid metrics: %s", invalid_metric_name, sorted(valid_metrics)
)
#########################################################
# End of pretty print functions
@ -1078,9 +1076,10 @@ class PrometheusLogger(CustomLogger):
except ImportError:
# Fallback to simple logging if rich is not available
verbose_logger.info(
f"Enabled metrics: {sorted(self.enabled_metrics) if hasattr(self, 'enabled_metrics') else 'All metrics'}"
"Enabled metrics: %s",
sorted(self.enabled_metrics) if hasattr(self, "enabled_metrics") else "All metrics",
)
verbose_logger.info(f"Label filters: {label_filters}")
verbose_logger.info("Label filters: %s", label_filters)
def _is_metric_enabled(self, metric_name: str) -> bool:
"""Check if a metric is enabled based on configuration"""
@ -1866,7 +1865,9 @@ class PrometheusLogger(CustomLogger):
for i, r in enumerate(results):
if isinstance(r, Exception):
verbose_logger.debug(
f"[Non-Blocking] Prometheus: Budget metric lookup {['key', 'team', 'user', 'org'][i]} failed: {r}"
"[Non-Blocking] Prometheus: Budget metric lookup %s failed: %s",
["key", "team", "user", "org"][i],
r,
)
def _increment_top_level_request_and_spend_metrics(
@ -2132,7 +2133,7 @@ class PrometheusLogger(CustomLogger):
response_cost=0,
)
except Exception as e:
verbose_logger.exception(f"prometheus Layer Error(): Exception occured - {e!s}")
verbose_logger.exception("prometheus Layer Error(): Exception occured - %s", e)
def _extract_status_code(
self,
@ -2262,8 +2263,9 @@ class PrometheusLogger(CustomLogger):
if self._is_invalid_api_key_request(status_code, exception=exception):
verbose_logger.debug(
"Skipping Prometheus metrics for invalid API key request: "
f"status_code={status_code}, exception={type(exception).__name__ if exception else None}"
"Skipping Prometheus metrics for invalid API key request: status_code=%s, exception=%s",
status_code,
type(exception).__name__ if exception else None,
)
return True
@ -2383,7 +2385,7 @@ class PrometheusLogger(CustomLogger):
)
except Exception as e:
verbose_logger.exception(f"prometheus Layer Error(): Exception occured - {e!s}")
verbose_logger.exception("prometheus Layer Error(): Exception occured - %s", e)
async def async_post_call_success_hook(self, data: dict, user_api_key_dict: UserAPIKeyAuth, response):
"""
@ -2608,7 +2610,7 @@ class PrometheusLogger(CustomLogger):
)
except Exception as e:
verbose_logger.debug(f"Prometheus Error: set_llm_deployment_failure_metrics. Exception occured - {e!s}")
verbose_logger.debug("Prometheus Error: set_llm_deployment_failure_metrics. Exception occured - %s", e)
def _set_deployment_tpm_rpm_limit_metrics(
self,
@ -2722,9 +2724,7 @@ class PrometheusLogger(CustomLogger):
)
self.litellm_remaining_requests_metric.labels(**_labels).set(remaining_requests)
except Exception as e:
verbose_logger.exception(
f"Prometheus Error: _async_set_router_remaining_metrics. Exception occured - {e!s}"
)
verbose_logger.exception("Prometheus Error: _async_set_router_remaining_metrics. Exception occured - %s", e)
def set_llm_deployment_success_metrics(
self,
@ -2867,7 +2867,7 @@ class PrometheusLogger(CustomLogger):
self.litellm_deployment_latency_per_output_token.labels(**_labels).observe(latency_per_token)
except Exception as e:
verbose_logger.exception(f"Prometheus Error: set_llm_deployment_success_metrics. Exception occured - {e!s}")
verbose_logger.exception("Prometheus Error: set_llm_deployment_success_metrics. Exception occured - %s", e)
return
def _record_guardrail_metrics(
@ -2912,7 +2912,7 @@ class PrometheusLogger(CustomLogger):
hook_type=hook_type,
).inc()
except Exception as e:
verbose_logger.debug(f"Error recording guardrail metrics: {e!s}")
verbose_logger.debug("Error recording guardrail metrics: %s", e)
########################################
# Managed Batch Metric Recording Methods
@ -2935,7 +2935,7 @@ class PrometheusLogger(CustomLogger):
api_key_alias=api_key_alias,
).inc()
except Exception as e:
verbose_logger.warning(f"Error recording batch created metric: {e}")
verbose_logger.warning("Error recording batch created metric: %s", e)
def record_managed_file_size(
self,
@ -2956,7 +2956,7 @@ class PrometheusLogger(CustomLogger):
user=user or "",
).set(size_bytes)
except Exception as e:
verbose_logger.warning(f"Error recording file size metric: {e}")
verbose_logger.warning("Error recording file size metric: %s", e)
def record_managed_batch_duration(
self,
@ -2970,7 +2970,7 @@ class PrometheusLogger(CustomLogger):
api_provider=api_provider or "",
).observe(duration_seconds)
except Exception as e:
verbose_logger.warning(f"Error recording batch duration metric: {e}")
verbose_logger.warning("Error recording batch duration metric: %s", e)
def record_managed_file_created(
self,
@ -2989,14 +2989,14 @@ class PrometheusLogger(CustomLogger):
api_key_alias=api_key_alias,
).inc()
except Exception as e:
verbose_logger.warning(f"Error recording file created metric: {e}")
verbose_logger.warning("Error recording file created metric: %s", e)
def record_managed_file_deleted(self, result: str):
"""Record a managed file deletion attempt. result is 'success' or 'blocked'."""
try:
self.litellm_managed_file_deleted_total.labels(result=result).inc()
except Exception as e:
verbose_logger.warning(f"Error recording file deleted metric: {e}")
verbose_logger.warning("Error recording file deleted metric: %s", e)
def record_check_batch_cost_run(
self,
@ -3023,7 +3023,7 @@ class PrometheusLogger(CustomLogger):
api_provider=api_provider or "",
).inc()
except Exception as e:
verbose_logger.warning(f"Error recording check batch cost metrics: {e}")
verbose_logger.warning("Error recording check batch cost metrics: %s", e)
def record_check_batch_cost_error(self, error_type: str):
try:
@ -3031,7 +3031,7 @@ class PrometheusLogger(CustomLogger):
error_type=error_type,
).inc()
except Exception as e:
verbose_logger.warning(f"Error recording check batch cost error metric: {e}")
verbose_logger.warning("Error recording check batch cost error metric: %s", e)
@staticmethod
def _get_exception_class_name(exception: Exception) -> str:
@ -3315,7 +3315,7 @@ class PrometheusLogger(CustomLogger):
await set_metrics_function(data)
except Exception as e:
verbose_logger.exception(f"Error initializing {data_type} budget metrics: {e!s}")
verbose_logger.exception("Error initializing %s budget metrics: %s", data_type, e)
async def _initialize_team_budget_metrics(self):
"""
@ -3495,18 +3495,18 @@ class PrometheusLogger(CustomLogger):
# Get total user count
total_users = await UserRepository(prisma_client).table.count()
self.litellm_total_users_metric.set(total_users)
verbose_logger.debug(f"Prometheus: set litellm_total_users to {total_users}")
verbose_logger.debug("Prometheus: set litellm_total_users to %s", total_users)
billable_users = await UserRepository(prisma_client).count_billable_users()
self.litellm_active_users_metric.set(billable_users)
verbose_logger.debug(f"Prometheus: set litellm_active_users to {billable_users}")
verbose_logger.debug("Prometheus: set litellm_active_users to %s", billable_users)
# Get total team count
total_teams = await TeamRepository(prisma_client).table.count()
self.litellm_teams_count_metric.set(total_teams)
verbose_logger.debug(f"Prometheus: set litellm_teams_count to {total_teams}")
verbose_logger.debug("Prometheus: set litellm_teams_count to %s", total_teams)
except Exception as e:
verbose_logger.exception(f"Error initializing user/team count metrics: {e!s}")
verbose_logger.exception("Error initializing user/team count metrics: %s", e)
async def _set_key_list_budget_metrics(self, keys: list[str | UserAPIKeyAuth]):
"""Helper function to set budget metrics for a list of keys"""
@ -3597,7 +3597,7 @@ class PrometheusLogger(CustomLogger):
user_api_key_cache=user_api_key_cache,
)
except Exception as e:
verbose_logger.debug(f"[Non-Blocking] Prometheus: Error getting team info: {e!s}")
verbose_logger.debug("[Non-Blocking] Prometheus: Error getting team info: %s", e)
return team_object
if team_info:
@ -3695,7 +3695,7 @@ class PrometheusLogger(CustomLogger):
include_budget_table=True,
)
except Exception as e:
verbose_logger.debug(f"[Non-Blocking] Prometheus: Error getting org info: {e!s}")
verbose_logger.debug("[Non-Blocking] Prometheus: Error getting org info: %s", e)
return
if org_info is None:
@ -3852,7 +3852,7 @@ class PrometheusLogger(CustomLogger):
if key_object:
user_api_key_dict.budget_reset_at = key_object.budget_reset_at
except Exception as e:
verbose_logger.debug(f"[Non-Blocking] Prometheus: Error getting key info: {e!s}")
verbose_logger.debug("[Non-Blocking] Prometheus: Error getting key info: %s", e)
return user_api_key_dict
@ -3917,7 +3917,7 @@ class PrometheusLogger(CustomLogger):
check_db_only=False,
)
except Exception as e:
verbose_logger.debug(f"[Non-Blocking] Prometheus: Error getting user info: {e!s}")
verbose_logger.debug("[Non-Blocking] Prometheus: Error getting user info: %s", e)
return user_object
if user_info:
@ -4174,44 +4174,6 @@ def get_custom_labels_from_metadata(metadata: dict) -> dict[str, str]:
return result
def get_service_tier_from_standard_logging_payload(
standard_logging_payload: StandardLoggingPayload,
) -> str | None:
"""
Resolve the service tier a request ran on, for the ``service_tier`` label.
The tier the provider actually served wins over the tier the caller asked for,
so latency and spend stay segmentable when the request said ``auto`` and the
provider picked the concrete tier. Providers report the served tier either at
the top level of the response (OpenAI, Bedrock, Groq) or on the usage object
(Anthropic).
Streaming responses carry no served tier, so the requested tier is the
fallback. That value is caller-controlled and survives param mapping even
where the provider then ignores it (Bedrock and Groq accept the request and
drop an unrecognized tier), so it is only labelled when it names a known
tier; otherwise one caller could mint a Prometheus series per string. Values
the provider itself reports are not caller-controlled and stay unrestricted,
so a tier a provider adds later is still labelled correctly.
"""
response = standard_logging_payload.get("response")
usage_object = standard_logging_payload.get("metadata", {}).get("usage_object")
served_candidates: tuple[object, ...] = (
response.get("service_tier") if isinstance(response, dict) else None,
usage_object.get("service_tier") if isinstance(usage_object, dict) else None,
)
served_tier = next((tier for tier in served_candidates if isinstance(tier, str) and tier), None)
if served_tier is not None:
return served_tier
model_parameters = standard_logging_payload.get("model_parameters")
requested_tier = model_parameters.get("service_tier") if isinstance(model_parameters, dict) else None
if isinstance(requested_tier, str) and requested_tier in KNOWN_REQUEST_SERVICE_TIERS:
return requested_tier
return None
def _get_combined_custom_metadata_from_standard_logging_payload(
standard_logging_payload: dict | None,
) -> dict[str, Any]:

View file

@ -82,7 +82,7 @@ class PrometheusServicesLogger:
self.mock_testing_failure_calls = 0
except Exception as e:
print_verbose(f"Got exception on init prometheus client {e!s}")
print_verbose(f"Got exception on init prometheus client {e}")
raise e
def _get_service_metrics_initialize(self, service: ServiceTypes) -> list[ServiceMetrics]:
@ -92,7 +92,7 @@ class PrometheusServicesLogger:
metrics = DEFAULT_SERVICE_CONFIGS.get(service, {}).get("metrics", [])
if not metrics:
verbose_logger.debug(f"No metrics found for service {service}")
verbose_logger.debug("No metrics found for service %s", service)
return DEFAULT_METRICS
return metrics

File diff suppressed because it is too large Load diff

View file

@ -31,7 +31,7 @@ class S3Logger:
import boto3
try:
verbose_logger.debug(f"in init s3 logger - s3_callback_params {litellm.s3_callback_params}")
verbose_logger.debug("in init s3 logger - s3_callback_params %s", litellm.s3_callback_params)
s3_use_team_prefix = False
@ -62,7 +62,7 @@ class S3Logger:
self.s3_server_side_encryption, self.s3_sse_kms_key_id = resolve_sse_params(
s3_server_side_encryption, s3_sse_kms_key_id
)
verbose_logger.debug(f"s3 logger using endpoint url {s3_endpoint_url}")
verbose_logger.debug("s3 logger using endpoint url %s", s3_endpoint_url)
# Create an S3 client with custom endpoint URL
self.s3_client = boto3.client(
"s3",
@ -78,7 +78,7 @@ class S3Logger:
**kwargs,
)
except Exception as e:
print_verbose(f"Got exception on init s3 client {e!s}")
print_verbose(f"Got exception on init s3 client {e}")
raise e
async def _async_log_event(self, kwargs, response_obj, start_time, end_time, print_verbose):
@ -86,7 +86,7 @@ class S3Logger:
def log_event(self, kwargs, response_obj, start_time, end_time, print_verbose):
try:
verbose_logger.debug(f"s3 Logging - Enters logging function for model {kwargs}")
verbose_logger.debug("s3 Logging - Enters logging function for model %s", kwargs)
# construct payload to send to s3
# follows the same params as langfuse.py
@ -163,19 +163,19 @@ class S3Logger:
**sse_params,
)
print_verbose(f"Response from s3:{response!s}")
print_verbose(f"Response from s3:{response}")
print_verbose(f"s3 Layer Logging - final response object: {response_obj}")
return response
except Exception as e:
verbose_logger.exception(f"s3 Layer Error - {e!s}")
verbose_logger.exception("s3 Layer Error - %s", e)
def _validated_sse_value(name: str, value: str | None) -> str | None:
if value is None or isinstance(value, str):
return value
verbose_logger.warning(
f"s3 logging: ignoring {name} because it has invalid type {type(value).__name__}; expected a string"
"s3 logging: ignoring %s because it has invalid type %s; expected a string", name, type(value).__name__
)
return None
@ -191,8 +191,8 @@ def resolve_sse_params(
return None, None
if valid_key_id and not algorithm.startswith("aws:kms"):
verbose_logger.warning(
f"s3 logging: ignoring s3_sse_kms_key_id because s3_server_side_encryption is {algorithm}; "
"set it to aws:kms to encrypt with the KMS key"
"s3 logging: ignoring s3_sse_kms_key_id because s3_server_side_encryption is %s; set it to aws:kms to encrypt with the KMS key",
algorithm,
)
return algorithm, None
return algorithm, valid_key_id

View file

@ -64,12 +64,12 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
_masker = SensitiveDataMasker()
if s3_callback_params_override is not None:
verbose_logger.debug(
f"in init s3 logger (audit override) - {_masker.mask_dict(dict(s3_callback_params_override))}"
"in init s3 logger (audit override) - %s", _masker.mask_dict(dict(s3_callback_params_override))
)
else:
verbose_logger.debug(
f"in init s3 logger - s3_callback_params "
f"{_masker.mask_dict(dict(litellm.s3_callback_params or {}))}"
"in init s3 logger - s3_callback_params %s",
_masker.mask_dict(dict(litellm.s3_callback_params or {})),
)
# Initialize S3 params first to get the correct s3_verify value
@ -98,11 +98,11 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
s3_server_side_encryption=s3_server_side_encryption,
s3_sse_kms_key_id=s3_sse_kms_key_id,
)
verbose_logger.debug(f"s3 logger using endpoint url {s3_endpoint_url}")
verbose_logger.debug("s3 logger using endpoint url %s", s3_endpoint_url)
# IMPORTANT
# Create httpx client AFTER _init_s3_params so we have the correct s3_verify value
verbose_logger.debug(f"s3_v2 logger creating async httpx client with s3_verify={self.s3_verify}")
verbose_logger.debug("s3_v2 logger creating async httpx client with s3_verify=%s", self.s3_verify)
self.async_httpx_client = get_async_httpx_client(
llm_provider=httpxSpecialProvider.LoggingCallback,
params={"ssl_verify": self.s3_verify},
@ -111,7 +111,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
asyncio.create_task(self.periodic_flush())
self.flush_lock = asyncio.Lock()
verbose_logger.debug(f"s3 flush interval: {s3_flush_interval}, s3 batch size: {s3_batch_size}")
verbose_logger.debug("s3 flush interval: %s, s3 batch size: %s", s3_flush_interval, s3_batch_size)
# Call CustomLogger's __init__
CustomBatchLogger.__init__(
self,
@ -125,7 +125,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
BaseAWSLLM.__init__(self)
except Exception as e:
print_verbose(f"Got exception on init s3 client {e!s}")
print_verbose(f"Got exception on init s3 client {e}")
raise e
def _init_s3_params(
@ -259,7 +259,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
async def _async_log_event_base(self, kwargs, response_obj, start_time, end_time):
try:
verbose_logger.debug(f"s3 Logging - Enters logging function for model {kwargs}")
verbose_logger.debug("s3 Logging - Enters logging function for model %s", kwargs)
s3_batch_logging_element = self.create_s3_batch_logging_element(
start_time=start_time,
@ -284,7 +284,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
self.batch_size,
)
except Exception as e:
verbose_logger.exception(f"s3 Layer Error - {e!s}")
verbose_logger.exception("s3 Layer Error - %s", e)
self.handle_callback_failure(callback_name="S3Logger")
async def async_upload_data_to_s3(self, batch_logging_element: s3BatchLoggingElement):
@ -313,8 +313,8 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
aws_sts_endpoint=self.s3_aws_sts_endpoint,
)
verbose_logger.debug(f"s3_v2 logger - uploading data to s3 - {batch_logging_element.s3_object_key}")
verbose_logger.debug(f"s3_v2 logger - s3_verify setting: {self.s3_verify}")
verbose_logger.debug("s3_v2 logger - uploading data to s3 - %s", batch_logging_element.s3_object_key)
verbose_logger.debug("s3_v2 logger - s3_verify setting: %s", self.s3_verify)
# Prepare the URL
url = f"https://{self.s3_bucket_name}.s3.{self.s3_region_name}.amazonaws.com/{batch_logging_element.s3_object_key}"
@ -374,16 +374,19 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
if response.status_code in (500, 503) and attempt < max_retries - 1:
wait_time = 2**attempt # 1s, 2s
verbose_logger.warning(
f"S3 upload returned {response.status_code}, retrying in {wait_time}s "
f"(attempt {attempt + 1}/{max_retries}) "
f"key={batch_logging_element.s3_object_key}"
"S3 upload returned %s, retrying in %ss (attempt %s/%s) key=%s",
response.status_code,
wait_time,
attempt + 1,
max_retries,
batch_logging_element.s3_object_key,
)
await asyncio.sleep(wait_time)
continue
response.raise_for_status()
break
except Exception as e:
verbose_logger.exception(f"Error uploading to s3: {e!s}")
verbose_logger.exception("Error uploading to s3: %s", e)
self.handle_callback_failure(callback_name="S3Logger")
async def async_send_batch(self):
@ -395,7 +398,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
Raises: Does not raise an exception, will only verbose_logger.exception()
"""
verbose_logger.debug(f"s3_v2 logger - sending batch of {len(self.log_queue)}")
verbose_logger.debug("s3_v2 logger - sending batch of %s", len(self.log_queue))
if not self.log_queue:
return
@ -447,7 +450,10 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
s3_file_name = litellm.utils.get_logging_id(start_time, standard_logging_payload) or ""
verbose_logger.debug(
f"Creating s3 file with prefix_components={prefix_components},prefix_path={prefix_path} and {s3_file_name}"
"Creating s3 file with prefix_components=%s,prefix_path=%s and %s",
prefix_components,
prefix_path,
s3_file_name,
)
s3_object_key = get_s3_object_key(
s3_path=cast(str | None, self.s3_path) or "",
@ -455,7 +461,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
start_time=start_time,
s3_file_name=s3_file_name,
)
verbose_logger.debug(f"s3_object_key={s3_object_key}")
verbose_logger.debug("s3_object_key=%s", s3_object_key)
s3_object_download_filename = (
f"time-{start_time.strftime('%Y-%m-%dT%H-%M-%S-%f')}_{standard_logging_payload['id']}.json"
@ -479,7 +485,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
except ImportError:
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
try:
verbose_logger.debug(f"s3_v2 logger - uploading data to s3 - {batch_logging_element.s3_object_key}")
verbose_logger.debug("s3_v2 logger - uploading data to s3 - %s", batch_logging_element.s3_object_key)
credentials: Credentials = self.get_credentials(
aws_access_key_id=self.s3_aws_access_key_id,
aws_secret_access_key=self.s3_aws_secret_access_key,
@ -548,16 +554,19 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
if response.status_code in (500, 503) and attempt < max_retries - 1:
wait_time = 2**attempt # 1s, 2s
verbose_logger.warning(
f"S3 upload returned {response.status_code}, retrying in {wait_time}s "
f"(attempt {attempt + 1}/{max_retries}) "
f"key={batch_logging_element.s3_object_key}"
"S3 upload returned %s, retrying in %ss (attempt %s/%s) key=%s",
response.status_code,
wait_time,
attempt + 1,
max_retries,
batch_logging_element.s3_object_key,
)
time.sleep(wait_time)
continue
response.raise_for_status()
break
except Exception as e:
verbose_logger.exception(f"Error uploading to s3: {e!s}")
verbose_logger.exception("Error uploading to s3: %s", e)
self.handle_callback_failure(callback_name="S3Logger")
async def _download_object_from_s3(self, s3_object_key: str) -> dict | None:
@ -596,7 +605,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
aws_sts_endpoint=self.s3_aws_sts_endpoint,
)
verbose_logger.debug(f"s3_v2 logger - downloading data from s3 - {s3_object_key}")
verbose_logger.debug("s3_v2 logger - downloading data from s3 - %s", s3_object_key)
# Prepare the URL
url = f"https://{self.s3_bucket_name}.s3.{self.s3_region_name}.amazonaws.com/{s3_object_key}"
@ -642,7 +651,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
return response.json()
except Exception as e:
verbose_logger.exception(f"Error downloading from S3: {e!s}")
verbose_logger.exception("Error downloading from S3: %s", e)
return None
async def get_proxy_server_request_from_cold_storage_with_object_key(
@ -666,5 +675,5 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM):
downloaded_object = await self._download_object_from_s3(object_key)
return downloaded_object
except Exception as e:
verbose_logger.exception(f"Error retrieving object {object_key} from cold storage: {e!s}")
verbose_logger.exception("Error retrieving object %s from cold storage: %s", object_key, e)
return None

View file

@ -68,7 +68,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
**kwargs,
) -> None:
try:
verbose_logger.debug(f"in init sqs logger - sqs_callback_params {litellm.aws_sqs_callback_params}")
verbose_logger.debug("in init sqs logger - sqs_callback_params %s", litellm.aws_sqs_callback_params)
self.async_httpx_client = get_async_httpx_client(
llm_provider=httpxSpecialProvider.LoggingCallback,
@ -100,7 +100,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
asyncio.create_task(self.periodic_flush())
self.flush_lock = asyncio.Lock()
verbose_logger.debug(f"sqs flush interval: {sqs_flush_interval}, sqs batch size: {sqs_batch_size}")
verbose_logger.debug("sqs flush interval: %s, sqs batch size: %s", sqs_flush_interval, sqs_batch_size)
CustomBatchLogger.__init__(
self,
@ -113,7 +113,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
BaseAWSLLM.__init__(self)
except Exception as e:
print_verbose(f"Got exception on init sqs client {e!s}")
print_verbose(f"Got exception on init sqs client {e}")
raise e
def _init_sqs_params(
@ -215,7 +215,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
self.batch_size,
)
except Exception as e:
verbose_logger.exception(f"sqs Layer Error - {e!s}")
verbose_logger.exception("sqs Layer Error - %s", e)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
try:
@ -233,10 +233,10 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
)
except Exception as e:
verbose_logger.exception(f"Datadog Layer Error - {e!s}\n{traceback.format_exc()}")
verbose_logger.exception("Datadog Layer Error - %s\n%s", e, traceback.format_exc())
async def async_send_batch(self) -> None:
verbose_logger.debug(f"sqs logger - sending batch of {len(self.log_queue)}")
verbose_logger.debug("sqs logger - sending batch of %s", len(self.log_queue))
if not self.log_queue:
return
@ -305,7 +305,7 @@ class SQSLogger(CustomBatchLogger, BaseAWSLLM):
)
response.raise_for_status()
except Exception as e:
verbose_logger.exception(f"Error sending to SQS: {e!s}")
verbose_logger.exception("Error sending to SQS: %s", e)
async def async_health_check(self) -> IntegrationHealthCheckStatus:
"""

View file

@ -15,7 +15,9 @@ class TraceloopLogger:
from traceloop.sdk.tracing.tracing import TracerWrapper
except ModuleNotFoundError as e:
verbose_logger.error(
f"Traceloop not installed, try running 'pip install traceloop-sdk' to fix this error: {e}\n{traceback.format_exc()}"
"Traceloop not installed, try running 'pip install traceloop-sdk' to fix this error: %s\n%s",
e,
traceback.format_exc(),
)
raise e

View file

@ -124,7 +124,7 @@ class VectorStorePreCallHook(CustomLogger):
},
)
verbose_logger.debug(f"search_response: {search_response}")
verbose_logger.debug("search_response: %s", search_response)
# Store search results for later use in citations
all_search_results.append(search_response)
@ -137,7 +137,7 @@ class VectorStorePreCallHook(CustomLogger):
# Get the number of results for logging
num_results = 0
num_results = len(search_response.get("data", []) or [])
verbose_logger.debug(f"Vector store search completed. Added context from {num_results} results")
verbose_logger.debug("Vector store search completed. Added context from %s results", num_results)
# Store search results as-is (already in OpenAI-compatible format)
if litellm_logging_obj and all_search_results:
@ -146,7 +146,7 @@ class VectorStorePreCallHook(CustomLogger):
return model, modified_messages, non_default_params
except Exception as e:
verbose_logger.exception(f"Error in VectorStorePreCallHook: {e!s}")
verbose_logger.exception("Error in VectorStorePreCallHook: %s", e)
# Return original parameters on error
return model, messages, non_default_params
@ -243,14 +243,14 @@ class VectorStorePreCallHook(CustomLogger):
verbose_logger.debug("No litellm_logging_obj in request_data")
return None
verbose_logger.debug(f"model_call_details keys: {list(litellm_logging_obj.model_call_details.keys())}")
verbose_logger.debug("model_call_details keys: %s", list(litellm_logging_obj.model_call_details.keys()))
# Get search results from model_call_details (already in OpenAI format)
search_results: list[VectorStoreSearchResponse] | None = litellm_logging_obj.model_call_details.get(
"search_results"
)
verbose_logger.debug(f"Search results found: {search_results is not None}")
verbose_logger.debug("Search results found: %s", search_results is not None)
if not search_results:
verbose_logger.debug("No search results found")
@ -269,13 +269,13 @@ class VectorStorePreCallHook(CustomLogger):
# Set the provider_specific_fields
setattr(choice.message, "provider_specific_fields", provider_fields)
verbose_logger.debug(f"Added {len(search_results)} search results to response")
verbose_logger.debug("Added %s search results to response", len(search_results))
# Return modified response
return response
except Exception as e:
verbose_logger.exception(f"Error adding search results to response: {e!s}")
verbose_logger.exception("Error adding search results to response: %s", e)
# Don't fail the request if search results fail to be added
return None
@ -297,7 +297,7 @@ class VectorStorePreCallHook(CustomLogger):
# Get search results from model_call_details (already in OpenAI format)
search_results: list[VectorStoreSearchResponse] | None = request_data.get("search_results")
verbose_logger.debug(f"Search results found for streaming chunk: {search_results is not None}")
verbose_logger.debug("Search results found for streaming chunk: %s", search_results is not None)
if not search_results:
verbose_logger.debug("No search results found for streaming chunk")
@ -316,12 +316,12 @@ class VectorStorePreCallHook(CustomLogger):
# Set the provider_specific_fields
choice.delta.provider_specific_fields = provider_fields
verbose_logger.debug(f"Added {len(search_results)} search results to streaming chunk")
verbose_logger.debug("Added %s search results to streaming chunk", len(search_results))
# Return modified chunk
return response_chunk
except Exception as e:
verbose_logger.exception(f"Error adding search results to streaming chunk: {e!s}")
verbose_logger.exception("Error adding search results to streaming chunk: %s", e)
# Don't fail the request if search results fail to be added
return response_chunk

View file

@ -148,10 +148,10 @@ def get_weave_otel_config() -> WeaveOtelConfig:
host = "https://" + host
# Self-managed instances use a different path
endpoint = host.rstrip("/") + WEAVE_OTEL_ENDPOINT
verbose_logger.debug(f"Using Weave OTEL endpoint from host: {endpoint}")
verbose_logger.debug("Using Weave OTEL endpoint from host: %s", endpoint)
else:
endpoint = WEAVE_BASE_URL + WEAVE_OTEL_ENDPOINT
verbose_logger.debug(f"Using Weave cloud endpoint: {endpoint}")
verbose_logger.debug("Using Weave cloud endpoint: %s", endpoint)
# Weave uses Basic auth with format: api:<WANDB_API_KEY>
auth_header = _get_weave_authorization_header(api_key=api_key)

View file

@ -155,8 +155,8 @@ class WebSearchInterceptionLogger(CustomLogger):
)
if anthropic_config is not None and anthropic_config.handles_web_search_natively():
verbose_logger.debug(
f"WebSearchInterception: Skipping short-circuit for {provider_str} "
"(provider handles web search natively via the agentic loop)"
"WebSearchInterception: Skipping short-circuit for %s (provider handles web search natively via the agentic loop)",
provider_str,
)
return None
except (ValueError, Exception):
@ -176,7 +176,7 @@ class WebSearchInterceptionLogger(CustomLogger):
return None
verbose_logger.debug(
f"WebSearchInterception: Short-circuit search detected (provider={provider_str}, query='{query}')"
"WebSearchInterception: Short-circuit search detected (provider=%s, query='%s')", provider_str, query
)
# Native clients (Claude Desktop / Cowork / Anthropic SDK) make a
@ -198,7 +198,7 @@ class WebSearchInterceptionLogger(CustomLogger):
else:
search_result_text, structured = await self._execute_search(query, kwargs=kwargs)
except Exception as e:
verbose_logger.error(f"WebSearchInterception: Short-circuit search failed: {e}")
verbose_logger.error("WebSearchInterception: Short-circuit search failed: %s", e)
search_result_text, structured = f"Search failed: {e}", None
content: list[dict[str, object]] = []
@ -224,7 +224,7 @@ class WebSearchInterceptionLogger(CustomLogger):
content.append({"type": "text", "text": search_result_text})
response: dict[str, object] = {
"id": f"msg_{uuid.uuid4()!s}",
"id": f"msg_{uuid.uuid4()}",
"type": "message",
"role": "assistant",
"model": model,
@ -235,9 +235,9 @@ class WebSearchInterceptionLogger(CustomLogger):
}
verbose_logger.debug(
"WebSearchInterception: Short-circuit search completed, "
f"returning synthetic response ({len(search_result_text)} chars, "
f"native_blocks={native_tool is not None})"
"WebSearchInterception: Short-circuit search completed, returning synthetic response (%s chars, native_blocks=%s)",
len(search_result_text),
native_tool is not None,
)
return response
@ -294,8 +294,10 @@ class WebSearchInterceptionLogger(CustomLogger):
converted_tool = get_litellm_web_search_tool_openai()
converted_tools.append(converted_tool)
verbose_logger.debug(
f"WebSearchInterception: Converted {tool.get('name', 'unknown')} "
f"(type={tool.get('type', 'none')}) to {LITELLM_WEB_SEARCH_TOOL_NAME}"
"WebSearchInterception: Converted %s (type=%s) to %s",
tool.get("name", "unknown"),
tool.get("type", "none"),
LITELLM_WEB_SEARCH_TOOL_NAME,
)
else:
# Keep other tools as-is
@ -419,14 +421,14 @@ class WebSearchInterceptionLogger(CustomLogger):
custom_llm_provider = kwargs.get("litellm_params", {}).get("custom_llm_provider", "")
verbose_logger.debug(
f"WebSearchInterception: Pre-request hook called"
f" - custom_llm_provider={custom_llm_provider}"
f" - enabled_providers={self.enabled_providers or 'ALL'}"
"WebSearchInterception: Pre-request hook called - custom_llm_provider=%s - enabled_providers=%s",
custom_llm_provider,
self.enabled_providers or "ALL",
)
if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers:
verbose_logger.debug(
f"WebSearchInterception: Skipping - provider {custom_llm_provider} not in {self.enabled_providers}"
"WebSearchInterception: Skipping - provider %s not in %s", custom_llm_provider, self.enabled_providers
)
return None
@ -440,7 +442,7 @@ class WebSearchInterceptionLogger(CustomLogger):
if not has_websearch:
return None
verbose_logger.debug(f"WebSearchInterception: Pre-request hook triggered for provider={custom_llm_provider}")
verbose_logger.debug("WebSearchInterception: Pre-request hook triggered for provider=%s", custom_llm_provider)
# If the client sent an Anthropic-native web_search_* tool, mark the
# request so the agentic loop emits native web_search_tool_result
@ -457,15 +459,17 @@ class WebSearchInterceptionLogger(CustomLogger):
standard_tool = get_litellm_web_search_tool()
converted_tools.append(standard_tool)
verbose_logger.debug(
f"WebSearchInterception: Converted {tool.get('name', 'unknown')} "
f"(type={tool.get('type', 'none')}) to {LITELLM_WEB_SEARCH_TOOL_NAME}"
"WebSearchInterception: Converted %s (type=%s) to %s",
tool.get("name", "unknown"),
tool.get("type", "none"),
LITELLM_WEB_SEARCH_TOOL_NAME,
)
else:
converted_tools.append(tool)
kwargs["tools"] = converted_tools
verbose_logger.debug(
f"WebSearchInterception: Tools after conversion: {[t.get('name') for t in converted_tools]}"
"WebSearchInterception: Tools after conversion: %s", [t.get("name") for t in converted_tools]
)
if "tool_choice" in kwargs:
@ -511,15 +515,17 @@ class WebSearchInterceptionLogger(CustomLogger):
kwargs=kwargs,
)
verbose_logger.debug(f"WebSearchInterception: Hook called! provider={custom_llm_provider}, stream={stream}")
verbose_logger.debug(f"WebSearchInterception: Response type: {type(response)}")
verbose_logger.debug("WebSearchInterception: Hook called! provider=%s, stream=%s", custom_llm_provider, stream)
verbose_logger.debug("WebSearchInterception: Response type: %s", type(response))
# Check if provider should be intercepted
# Note: custom_llm_provider is already normalized by get_llm_provider()
# (e.g., "bedrock/invoke/..." -> "bedrock")
if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers:
verbose_logger.debug(
f"WebSearchInterception: Skipping provider {custom_llm_provider} (not in enabled list: {self.enabled_providers})"
"WebSearchInterception: Skipping provider %s (not in enabled list: %s)",
custom_llm_provider,
self.enabled_providers,
)
return False, {}
@ -541,7 +547,7 @@ class WebSearchInterceptionLogger(CustomLogger):
return False, {}
verbose_logger.debug(
f"WebSearchInterception: Detected {len(tool_calls)} WebSearch tool call(s), executing agentic loop"
"WebSearchInterception: Detected %s WebSearch tool call(s), executing agentic loop", len(tool_calls)
)
# Extract thinking blocks from response content.
@ -576,7 +582,7 @@ class WebSearchInterceptionLogger(CustomLogger):
if thinking_blocks:
verbose_logger.debug(
f"WebSearchInterception: Extracted {len(thinking_blocks)} thinking block(s) from response"
"WebSearchInterception: Extracted %s thinking block(s) from response", len(thinking_blocks)
)
# Return tools dict with tool calls and thinking blocks
@ -606,14 +612,16 @@ class WebSearchInterceptionLogger(CustomLogger):
"""
verbose_logger.debug(
f"WebSearchInterception: Chat completion hook called! provider={custom_llm_provider}, stream={stream}"
"WebSearchInterception: Chat completion hook called! provider=%s, stream=%s", custom_llm_provider, stream
)
verbose_logger.debug(f"WebSearchInterception: Response type: {type(response)}")
verbose_logger.debug("WebSearchInterception: Response type: %s", type(response))
# Check if provider should be intercepted
if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers:
verbose_logger.debug(
f"WebSearchInterception: Skipping provider {custom_llm_provider} (not in enabled list: {self.enabled_providers})"
"WebSearchInterception: Skipping provider %s (not in enabled list: %s)",
custom_llm_provider,
self.enabled_providers,
)
return False, {}
@ -635,7 +643,7 @@ class WebSearchInterceptionLogger(CustomLogger):
return False, {}
verbose_logger.debug(
f"WebSearchInterception: Detected {len(tool_calls)} WebSearch tool call(s), executing agentic loop"
"WebSearchInterception: Detected %s WebSearch tool call(s), executing agentic loop", len(tool_calls)
)
# Return tools dict with tool calls
@ -659,12 +667,14 @@ class WebSearchInterceptionLogger(CustomLogger):
) -> tuple[bool, dict]:
"""Check if WebSearch interception is needed for the Responses API."""
verbose_logger.debug(
f"WebSearchInterception: Responses hook called! provider={custom_llm_provider}, stream={stream}"
"WebSearchInterception: Responses hook called! provider=%s, stream=%s", custom_llm_provider, stream
)
if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers:
verbose_logger.debug(
f"WebSearchInterception: Skipping provider {custom_llm_provider} (not in enabled list: {self.enabled_providers})"
"WebSearchInterception: Skipping provider %s (not in enabled list: %s)",
custom_llm_provider,
self.enabled_providers,
)
return False, {}
@ -684,7 +694,7 @@ class WebSearchInterceptionLogger(CustomLogger):
return False, {}
verbose_logger.debug(
f"WebSearchInterception: Detected {len(tool_calls)} WebSearch function_call(s), executing agentic loop"
"WebSearchInterception: Detected %s WebSearch function_call(s), executing agentic loop", len(tool_calls)
)
tools_dict = {
@ -716,7 +726,7 @@ class WebSearchInterceptionLogger(CustomLogger):
tool_calls = tools["tool_calls"]
thinking_blocks = tools.get("thinking_blocks", [])
verbose_logger.debug(f"WebSearchInterception: Executing agentic loop for {len(tool_calls)} search(es)")
verbose_logger.debug("WebSearchInterception: Executing agentic loop for %s search(es)", len(tool_calls))
return await self._execute_agentic_loop(
model=model,
@ -853,7 +863,8 @@ class WebSearchInterceptionLogger(CustomLogger):
# Object refused write — fall through and leave the response
# untouched rather than crash the request.
verbose_logger.debug(
f"WebSearchInterception: could not inject native blocks into response of type {type(response).__name__}"
"WebSearchInterception: could not inject native blocks into response of type %s",
type(response).__name__,
)
return response
@ -878,7 +889,7 @@ class WebSearchInterceptionLogger(CustomLogger):
response_format = tools.get("response_format", "openai")
verbose_logger.debug(
f"WebSearchInterception: Executing chat completion agentic loop for {len(tool_calls)} search(es)"
"WebSearchInterception: Executing chat completion agentic loop for %s search(es)", len(tool_calls)
)
return await self._execute_chat_completion_agentic_loop(
@ -962,7 +973,7 @@ class WebSearchInterceptionLogger(CustomLogger):
for tool_call in tool_calls
]
verbose_logger.debug(f"WebSearchInterception: Executing {len(search_tasks)} responses search(es) in parallel")
verbose_logger.debug("WebSearchInterception: Executing %s responses search(es) in parallel", len(search_tasks))
search_results = await asyncio.gather(*search_tasks, return_exceptions=True)
search_texts = [self._extract_search_text(result) for result in search_results]
@ -1038,12 +1049,12 @@ class WebSearchInterceptionLogger(CustomLogger):
@staticmethod
def _extract_search_text(result: object) -> str:
if isinstance(result, Exception):
verbose_logger.error(f"WebSearchInterception: Responses search failed with error: {result!s}")
return f"Search failed: {result!s}"
verbose_logger.error("WebSearchInterception: Responses search failed with error: %s", result)
return f"Search failed: {result}"
if isinstance(result, tuple) and len(result) == 2:
text_value, _ = result
return text_value if isinstance(text_value, str) else str(text_value)
verbose_logger.debug(f"WebSearchInterception: Unexpected search result type {type(result)}")
verbose_logger.debug("WebSearchInterception: Unexpected search result type %s", type(result))
return str(result)
@staticmethod
@ -1176,15 +1187,15 @@ class WebSearchInterceptionLogger(CustomLogger):
for tool_call in tool_calls:
query = tool_call["input"].get("query")
if query:
verbose_logger.debug(f"WebSearchInterception: Queuing search for query='{query}'")
verbose_logger.debug("WebSearchInterception: Queuing search for query='%s'", query)
search_tasks.append(self._execute_search(query, kwargs=kwargs))
else:
verbose_logger.debug(f"WebSearchInterception: Tool call {tool_call['id']} has no query")
verbose_logger.debug("WebSearchInterception: Tool call %s has no query", tool_call["id"])
# Add empty result for tools without query
search_tasks.append(self._create_empty_search_result())
# Execute searches in parallel
verbose_logger.debug(f"WebSearchInterception: Executing {len(search_tasks)} search(es) in parallel")
verbose_logger.debug("WebSearchInterception: Executing %s search(es) in parallel", len(search_tasks))
search_results = await asyncio.gather(*search_tasks, return_exceptions=True)
# Split the gathered (text, structured) tuples into two parallel lists.
@ -1194,8 +1205,8 @@ class WebSearchInterceptionLogger(CustomLogger):
structured_results: list[SearchResponse | None] = []
for i, result in enumerate(search_results):
if isinstance(result, Exception):
verbose_logger.error(f"WebSearchInterception: Search {i} failed with error: {result!s}")
final_search_results.append(f"Search failed: {result!s}")
verbose_logger.error("WebSearchInterception: Search %s failed with error: %s", i, result)
final_search_results.append(f"Search failed: {result}")
structured_results.append(None)
elif isinstance(result, tuple) and len(result) == 2:
text_value, structured_value = result
@ -1204,7 +1215,7 @@ class WebSearchInterceptionLogger(CustomLogger):
else:
# Defensive: legacy callers / unexpected shape — preserve text,
# drop structure.
verbose_logger.debug(f"WebSearchInterception: Unexpected result type {type(result)} at index {i}")
verbose_logger.debug("WebSearchInterception: Unexpected result type %s at index %s", type(result), i)
final_search_results.append(str(result))
structured_results.append(None)
@ -1224,7 +1235,7 @@ class WebSearchInterceptionLogger(CustomLogger):
max_tokens = self._resolve_max_tokens(anthropic_messages_optional_request_params, kwargs)
verbose_logger.debug(f"WebSearchInterception: Using max_tokens={max_tokens} for follow-up request")
verbose_logger.debug("WebSearchInterception: Using max_tokens=%s for follow-up request", max_tokens)
optional_params_without_max_tokens = {
k: v for k, v in anthropic_messages_optional_request_params.items() if k != "max_tokens"
@ -1286,12 +1297,12 @@ class WebSearchInterceptionLogger(CustomLogger):
if not search_provider:
search_provider = "perplexity"
verbose_logger.debug(
"WebSearchInterception: No search tools configured in router, "
f"using default provider '{search_provider}'"
"WebSearchInterception: No search tools configured in router, using default provider '%s'",
search_provider,
)
verbose_logger.debug(
f"WebSearchInterception: Executing search for '{query}' using provider '{search_provider}'"
"WebSearchInterception: Executing search for '%s' using provider '%s'", query, search_provider
)
search_kwargs = {
key: value
@ -1304,11 +1315,11 @@ class WebSearchInterceptionLogger(CustomLogger):
search_result_text = WebSearchTransformation.format_search_response(result)
verbose_logger.debug(
f"WebSearchInterception: Search completed for '{query}', got {len(search_result_text)} chars"
"WebSearchInterception: Search completed for '%s', got %s chars", query, len(search_result_text)
)
return search_result_text, result
except Exception as e:
verbose_logger.error(f"WebSearchInterception: Search failed for '{query}': {e!s}")
verbose_logger.error("WebSearchInterception: Search failed for '%s': %s", query, e)
raise
async def _authorize_search_tool(
@ -1392,21 +1403,25 @@ class WebSearchInterceptionLogger(CustomLogger):
if matching_tools:
search_provider = (matching_tools[0].get("litellm_params", {}) or {}).get("search_provider")
verbose_logger.debug(
f"WebSearchInterception: Found search tool '{self.search_tool_name}' "
f"from {source} with provider '{search_provider}'"
"WebSearchInterception: Found search tool '%s' from %s with provider '%s'",
self.search_tool_name,
source,
search_provider,
)
return matching_tools[0]
verbose_logger.debug(
f"WebSearchInterception: Search tool '{self.search_tool_name}' not found in {source}, "
"falling back to first available or perplexity"
"WebSearchInterception: Search tool '%s' not found in %s, falling back to first available or perplexity",
self.search_tool_name,
source,
)
if search_tools:
first_tool = search_tools[0]
search_provider = (first_tool.get("litellm_params", {}) or {}).get("search_provider")
verbose_logger.debug(
f"WebSearchInterception: Using first available search tool from {source} "
f"with provider '{search_provider}'"
"WebSearchInterception: Using first available search tool from %s with provider '%s'",
source,
search_provider,
)
return first_tool
@ -1470,15 +1485,15 @@ class WebSearchInterceptionLogger(CustomLogger):
query = args.get("query")
if query:
verbose_logger.debug(f"WebSearchInterception: Queuing search for query='{query}'")
verbose_logger.debug("WebSearchInterception: Queuing search for query='%s'", query)
search_tasks.append(self._execute_search(query, kwargs=kwargs))
else:
verbose_logger.debug(f"WebSearchInterception: Tool call {tool_call.get('id')} has no query")
verbose_logger.debug("WebSearchInterception: Tool call %s has no query", tool_call.get("id"))
# Add empty result for tools without query
search_tasks.append(self._create_empty_search_result())
# Execute searches in parallel
verbose_logger.debug(f"WebSearchInterception: Executing {len(search_tasks)} search(es) in parallel")
verbose_logger.debug("WebSearchInterception: Executing %s search(es) in parallel", len(search_tasks))
search_results = await asyncio.gather(*search_tasks, return_exceptions=True)
# Chat-completion path only needs text — OpenAI tool_result format
@ -1486,13 +1501,13 @@ class WebSearchInterceptionLogger(CustomLogger):
final_search_results: list[str] = []
for i, result in enumerate(search_results):
if isinstance(result, Exception):
verbose_logger.error(f"WebSearchInterception: Search {i} failed with error: {result!s}")
final_search_results.append(f"Search failed: {result!s}")
verbose_logger.error("WebSearchInterception: Search %s failed with error: %s", i, result)
final_search_results.append(f"Search failed: {result}")
elif isinstance(result, tuple) and len(result) == 2:
text_value, _ = result
final_search_results.append(cast(str, text_value) if isinstance(text_value, str) else str(text_value))
else:
verbose_logger.debug(f"WebSearchInterception: Unexpected result type {type(result)} at index {i}")
verbose_logger.debug("WebSearchInterception: Unexpected result type %s at index %s", type(result), i)
final_search_results.append(str(result))
# Build assistant and tool messages using transformation
@ -1517,7 +1532,7 @@ class WebSearchInterceptionLogger(CustomLogger):
]
verbose_logger.debug("WebSearchInterception: Making follow-up chat completion request with search results")
verbose_logger.debug(f"WebSearchInterception: Follow-up messages count: {len(follow_up_messages)}")
verbose_logger.debug("WebSearchInterception: Follow-up messages count: %s", len(follow_up_messages))
# Remove internal parameters that shouldn't be passed to follow-up request
internal_params = {

View file

@ -103,7 +103,7 @@ class WebSearchTransformation:
parsed_input = json.loads(arguments) if arguments else {}
except json.JSONDecodeError:
verbose_logger.warning(
f"WebSearchInterception: Failed to parse function_call arguments: {arguments}"
"WebSearchInterception: Failed to parse function_call arguments: %s", arguments
)
parsed_input = {}
elif isinstance(arguments, dict):
@ -122,7 +122,7 @@ class WebSearchTransformation:
"input": parsed_input,
}
)
verbose_logger.debug(f"WebSearchInterception: Found {item_name} function_call with call_id={call_id}")
verbose_logger.debug("WebSearchInterception: Found %s function_call with call_id=%s", item_name, call_id)
return len(tool_calls) > 0, tool_calls
@ -178,7 +178,7 @@ class WebSearchTransformation:
"input": block_input,
}
tool_calls.append(tool_call)
verbose_logger.debug(f"WebSearchInterception: Found {block_name} tool_use with id={tool_call['id']}")
verbose_logger.debug("WebSearchInterception: Found %s tool_use with id=%s", block_name, tool_call["id"])
return len(tool_calls) > 0, tool_calls
@ -255,7 +255,7 @@ class WebSearchTransformation:
arguments = json.loads(function_arguments)
except json.JSONDecodeError:
verbose_logger.warning(
f"WebSearchInterception: Failed to parse function arguments: {function_arguments}"
"WebSearchInterception: Failed to parse function arguments: %s", function_arguments
)
arguments = {}
else:
@ -273,7 +273,7 @@ class WebSearchTransformation:
"input": arguments, # For compatibility with Anthropic format
}
tool_calls.append(tool_call_dict)
verbose_logger.debug(f"WebSearchInterception: Found {function_name} tool_call with id={tool_id}")
verbose_logger.debug("WebSearchInterception: Found %s tool_call with id=%s", function_name, tool_id)
return len(tool_calls) > 0, tool_calls

View file

@ -42,9 +42,9 @@ try:
elif response["object"] == "chat.completion":
return self._resolve_chat_completion(request, response, time_elapsed)
else:
logger.debug(f"Unknown OpenAI response object: {response['object']}")
logger.debug("Unknown OpenAI response object: %s", response["object"])
except Exception as e:
logger.warning(f"Failed to resolve request/response: {e}")
logger.warning("Failed to resolve request/response: %s", e)
return None
@staticmethod

Some files were not shown because too many files have changed in this diff Show more