Merge remote-tracking branch 'origin/main' into litellm_integration_messages_folder

This commit is contained in:
kerry 2026-09-26 22:31:53 +00:00
commit 2a7d223318
27 changed files with 1211 additions and 276 deletions

View file

@ -54,7 +54,7 @@ After: the same request comes back with real token counts, so the dashboard show
## Affected release
<!-- Only for a fix to a regression in a released or rc version (perf, memory, crash, or behavior): name the version it regressed in, e.g. "regression in v1.100.0" or "since v1.101.0-rc.1", and add the `backport-stable` label so the fix is cherry-picked onto the rc line before the stable is tagged. Drop the section otherwise -->
<!-- Only for a fix to a regression in a released or rc version (perf, memory, crash, or behavior): name the version it regressed in, e.g. "regression in v1.100.0" or "since v1.101.0-rc.1". Add the `backport-stable` label only when the regression is a P0, meaning its Linear ticket is Urgent (a security hole however narrow, data loss, or a crash or outage for every user on that version), because every labeled PR must be cherry-picked onto the baking rc line before the stable can be tagged; every other regression fix ships in the next rc unlabeled. Drop the section otherwise -->
## Linear ticket

View file

@ -1610,6 +1610,7 @@ ALLOWED_VERTEX_AI_PASSTHROUGH_HEADERS: Final = {
# e.g., 'x-pass-anthropic-beta: value' becomes 'anthropic-beta: value'
# Works for all LLM pass-through endpoints (Vertex AI, Anthropic, Bedrock, etc.)
PASS_THROUGH_HEADER_PREFIX: Final = "x-pass-"
INTERNAL_KWARG_PREFIX: Final = "_litellm_"
AZURE_SPEECH_CUSTOM_LLM_PROVIDER: Final = "azure_speech"
AZURE_SPEECH_PASS_THROUGH_ROUTE_PREFIX: Final = "/azure_speech"

View file

@ -52,7 +52,7 @@ from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import (
LITELLM_IMAGE_VARIATION_PROVIDERS,
LlmProviders,
all_litellm_params,
is_litellm_owned_kwarg,
)
from litellm.utils import (
ImageResponse,
@ -249,11 +249,9 @@ def image_generation(
"size",
"style",
]
litellm_params: Final = all_litellm_params
default_params: Final = openai_params + litellm_params
non_default_params: Final = {
k: v for k, v in kwargs.items() if k not in default_params
} # model-specific params - pass them straight to the model/provider
k: v for k, v in kwargs.items() if k not in openai_params and not is_litellm_owned_kwarg(k)
}
image_generation_config: BaseImageGenerationConfig | None = None
if custom_llm_provider is not None and custom_llm_provider in LlmProviders._member_map_.values():
@ -757,11 +755,9 @@ def image_edit(
"style",
"async_call",
]
litellm_params_list: Final = all_litellm_params
default_params: Final = openai_params + litellm_params_list
non_default_params: Final = {
k: v for k, v in kwargs.items() if k not in default_params
} # model-specific params - pass them straight to the model/provider
k: v for k, v in kwargs.items() if k not in openai_params and not is_litellm_owned_kwarg(k)
}
litellm_logging_obj: Final[LiteLLMLoggingObj] = kwargs.get("litellm_logging_obj")
litellm_call_id: Final[str | None] = kwargs.get("litellm_call_id", None)
model_info: Final = kwargs.get("model_info", None)

View file

@ -58,7 +58,7 @@ from litellm.types.llms.openai import (
OpenAIFileObject,
PathLike,
)
from litellm.types.utils import ExtractedFileData, LlmProviders, SpecialEnums, all_litellm_params
from litellm.types.utils import ExtractedFileData, LlmProviders, SpecialEnums, is_litellm_owned_kwarg
from litellm.utils import get_llm_provider, get_optional_params
from ..base_aws_llm import BaseAWSLLM
@ -907,7 +907,7 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
{
k: v
for k, v in optional_params.items()
if k not in all_litellm_params or k in _LITELLM_PARAMS_THE_MAPPER_TAKES
if not is_litellm_owned_kwarg(k) or k in _LITELLM_PARAMS_THE_MAPPER_TAKES
}
),
)

View file

@ -18,7 +18,7 @@ from litellm.llms.base_llm.text_to_speech.transformation import (
TextToSpeechRequestData,
)
from litellm.secret_managers.main import get_secret_str
from litellm.types.utils import all_litellm_params
from litellm.types.utils import is_litellm_owned_kwarg
from ..common_utils import ElevenLabsException
@ -241,7 +241,7 @@ class ElevenLabsTextToSpeechConfig(BaseTextToSpeechConfig):
continue
mapped_params[key] = value
reserved_kwarg_keys: Final = set(all_litellm_params) | {
reserved_kwarg_keys: Final = {
self.ELEVENLABS_QUERY_PARAMS_KEY,
self.ELEVENLABS_VOICE_ID_KEY,
"voice",
@ -260,7 +260,7 @@ class ElevenLabsTextToSpeechConfig(BaseTextToSpeechConfig):
mapped_params[key] = value
for key in list(kwargs.keys()):
if key in reserved_kwarg_keys:
if key in reserved_kwarg_keys or is_litellm_owned_kwarg(key):
continue
value = kwargs[key]
if value is None:

View file

@ -284,7 +284,7 @@ from .types.utils import (
LlmProviders,
PromptTokensDetails,
ProviderSpecificHeader,
all_litellm_params,
is_litellm_owned_kwarg,
)
####### ENVIRONMENT VARIABLES ###################
@ -6351,15 +6351,10 @@ def embedding(
"max_retries",
"encoding_format",
]
litellm_params: Final = [
"aembedding",
"extra_headers",
] + all_litellm_params
default_params: Final = openai_params + litellm_params
default_params: Final = [*openai_params, "aembedding", "extra_headers"]
non_default_params: Final = {
k: v for k, v in kwargs.items() if k not in default_params
} # model-specific params - pass them straight to the model/provider
k: v for k, v in kwargs.items() if k not in default_params and not is_litellm_owned_kwarg(k)
}
model, custom_llm_provider, dynamic_api_key, api_base = get_llm_provider(
model=model,

View file

@ -12216,7 +12216,8 @@
"text",
"image"
],
"supports_embedding_image_input": true
"supports_embedding_image_input": true,
"input_cost_per_image_token": 4.7e-07
},
"azure_ai/grok-4": {
"input_cost_per_token": 3e-06,
@ -15804,7 +15805,9 @@
"mode": "embedding",
"output_cost_per_token": 0.0,
"output_vector_size": 1536,
"supports_embedding_image_input": true
"supports_embedding_image_input": true,
"input_cost_per_image_token": 4.7e-07,
"source": "https://cohere.com/pricing"
},
"cohere/parse-v5.0": {
"litellm_provider": "cohere",
@ -41913,14 +41916,14 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v4-pro": {
"cache_read_input_token_cost": 3.828e-08,
"input_cost_per_token": 4.5936e-07,
"cache_read_input_token_cost": 2.9e-08,
"input_cost_per_token": 3.48e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
"max_tokens": 384000,
"mode": "chat",
"output_cost_per_token": 9.1872e-07,
"output_cost_per_token": 6.96e-07,
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
"supports_function_calling": true,
@ -41933,14 +41936,14 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v4.1-flash": {
"cache_read_input_token_cost": 4.2e-09,
"input_cost_per_token": 1.4e-07,
"cache_read_input_token_cost": 1e-09,
"input_cost_per_token": 3.5e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"max_output_tokens": 384000,
"max_tokens": 384000,
"mode": "chat",
"output_cost_per_token": 4.2e-07,
"output_cost_per_token": 2.9e-07,
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
"supports_function_calling": true,
@ -43052,7 +43055,6 @@
"supports_web_search": false
},
"openrouter/openai/gpt-oss-20b": {
"cache_read_input_token_cost": 3e-08,
"input_cost_per_token": 1.8e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 131072,
@ -43322,8 +43324,8 @@
"input_cost_per_token": 2.6e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 65536,
"max_tokens": 65536,
"max_output_tokens": 235929,
"max_tokens": 235929,
"mode": "chat",
"output_cost_per_token": 2.08e-06,
"source": "https://openrouter.ai/api/v1/models",
@ -43603,8 +43605,8 @@
"input_cost_per_token": 9.646e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 204800,
"max_output_tokens": 128000,
"max_tokens": 128000,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 3.0316e-06,
"source": "https://openrouter.ai/api/v1/models",
@ -66825,14 +66827,14 @@
"supports_web_search": false
},
"openrouter/z-ai/glm-5.3-flash": {
"cache_read_input_token_cost": 1e-08,
"input_cost_per_token": 4.5e-08,
"cache_read_input_token_cost": 1.5e-08,
"input_cost_per_token": 4e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1310720,
"max_output_tokens": 943718,
"max_tokens": 943718,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 1.4e-07,
"output_cost_per_token": 5e-07,
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
"supports_function_calling": true,
@ -66865,13 +66867,13 @@
"supports_web_search": false
},
"openrouter/z-ai/glm-5.3": {
"input_cost_per_token": 1.4e-06,
"output_cost_per_token": 4.4e-06,
"cache_read_input_token_cost": 2.6e-07,
"input_cost_per_token": 3.794e-07,
"output_cost_per_token": 1.1924e-06,
"cache_read_input_token_cost": 7.046e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1310720,
"max_output_tokens": 943717,
"max_tokens": 943717,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
@ -67520,8 +67522,8 @@
"input_cost_per_token": 3.2e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 262140,
"max_tokens": 262140,
"max_output_tokens": 81920,
"max_tokens": 81920,
"mode": "chat",
"output_cost_per_token": 3.2e-06,
"source": "https://openrouter.ai/api/v1/models",
@ -68297,8 +68299,8 @@
"input_cost_per_token": 1.5e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 32768,
"max_tokens": 32768,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 6e-07,
"source": "https://openrouter.ai/api/v1/models",
@ -68476,8 +68478,8 @@
"cache_read_input_token_cost": 7e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 16384,
"max_tokens": 16384,
"max_output_tokens": 235929,
"max_tokens": 235929,
"mode": "chat",
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
@ -68637,8 +68639,8 @@
"output_cost_per_token": 3e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 32000,
"max_tokens": 32000,
"max_output_tokens": 235929,
"max_tokens": 235929,
"mode": "chat",
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
@ -69502,7 +69504,9 @@
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 1.5e-07,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 128000,
"max_tokens": 128000
},
"vertex_ai/gemini-2.5-flash-native-audio": {
"deprecation_date": "2026-12-13",
@ -69613,42 +69617,72 @@
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 3.5e-06,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 4096,
"max_tokens": 4096
},
"together_ai/meta-llama/Llama-3.2-1B-Instruct": {
"input_cost_per_token": 6e-08,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 6e-08,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 131072,
"max_tokens": 131072
},
"together_ai/meta-llama/Llama-3.2-3B-Instruct": {
"input_cost_per_token": 6e-08,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 6e-08,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 131072,
"max_tokens": 131072
},
"together_ai/Qwen/Qwen2-1.5B-Instruct": {
"input_cost_per_token": 2e-08,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 2e-08,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 32768,
"max_tokens": 32768
},
"together_ai/Qwen/Qwen2.5-14B-Instruct": {
"input_cost_per_token": 8e-07,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 8e-07,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 32768,
"max_tokens": 32768
},
"together_ai/Qwen/Qwen2.5-72B-Instruct": {
"input_cost_per_token": 1.2e-06,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 32768,
"max_tokens": 32768
},
"together_ai/Salesforce/Llama-Rank-V1": {
"input_cost_per_token": 1e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 8192,
"max_tokens": 8192,
"mode": "rerank",
"output_cost_per_token": 0.0,
"source": "https://api.together.xyz/v1/models"
},
"together_ai/meta-llama/Meta-Llama-3.1-8B": {
"input_cost_per_token": 2e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 16384,
"max_tokens": 16384,
"mode": "completion",
"output_cost_per_token": 2e-07,
"source": "https://api.together.xyz/v1/models"
},
"together_ai/together/Tev1-4B-experimental": {
"cache_read_input_token_cost": 4.2e-08,

View file

@ -2983,6 +2983,14 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
"Set this well above health_check_interval because /health and the UI read the latest row per model."
),
)
maximum_daily_tag_spend_retention_period: str | None = Field(
None,
description=(
"Maximum retention period for per-day tag spend aggregate rows (e.g., '90d'). Rows whose day is older "
"than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rows are never "
"deleted. Only historical tag usage analytics are affected; tag budgets read the lifetime counter."
),
)
use_spend_logs_partitioning: bool | None = Field(
None,
description="If True and LiteLLM_SpendLogs has been converted to a range-partitioned table (db_scripts/partition_spend_logs.sql), retention cleanup drops expired partitions instead of deleting rows, and pre-creates upcoming partitions. Default is False.",

View file

@ -32,6 +32,17 @@ from litellm.proxy.utils import PrismaClient
StopReason: TypeAlias = Literal["exhausted", "budget_exhausted", "batch_cap_reached", "aborted"]
Cutoff: TypeAlias = datetime | str
"""Rows strictly older than this are expired: a timestamp, or an ISO calendar day for tables keyed by day"""
def _cutoff_cast(cutoff: Cutoff) -> str:
return "timestamptz" if isinstance(cutoff, datetime) else "text"
def _cutoff_text(cutoff: Cutoff) -> str:
return cutoff.isoformat() if isinstance(cutoff, datetime) else cutoff
@dataclass(frozen=True, slots=True)
class TableCleanupResult:
@ -278,7 +289,7 @@ class SpendLogCleanup:
return remaining
async def _execute_delete_batch(
self, prisma_client: PrismaClient, delete_sql: str, cutoff_date: datetime, deadline: float
self, prisma_client: PrismaClient, delete_sql: str, cutoff_date: Cutoff, deadline: float
) -> int | None:
"""
Run one delete batch under a Postgres statement and lock timeout.
@ -301,7 +312,7 @@ class SpendLogCleanup:
return deleted_result if isinstance(deleted_result, int) else None
async def _count_remaining(
self, prisma_client: PrismaClient, cutoff_date: datetime, table_name: str, time_column: str, deadline: float
self, prisma_client: PrismaClient, cutoff_date: Cutoff, table_name: str, time_column: str, deadline: float
) -> int | None:
"""
Count expired rows still outstanding, stopping at a cap.
@ -314,7 +325,7 @@ class SpendLogCleanup:
count_sql: Final = f"""
SELECT count(*)::int AS remaining FROM (
SELECT 1 FROM "{table_name}"
WHERE "{time_column}" < $1::timestamptz
WHERE "{time_column}" < $1::{_cutoff_cast(cutoff_date)}
LIMIT $2
) capped
"""
@ -332,7 +343,7 @@ class SpendLogCleanup:
async def _delete_old_rows_batched(
self,
prisma_client: PrismaClient,
cutoff_date: datetime,
cutoff_date: Cutoff,
table_name: str,
key_columns: tuple[str, ...],
time_column: str,
@ -350,7 +361,7 @@ class SpendLogCleanup:
DELETE FROM "{table_name}"
WHERE ({key_list}) IN (
SELECT {key_list} FROM "{table_name}"
WHERE "{time_column}" < $1::timestamptz
WHERE "{time_column}" < $1::{_cutoff_cast(cutoff_date)}
LIMIT $2
)
"""
@ -406,7 +417,7 @@ class SpendLogCleanup:
run_count,
consecutive_failures,
self.batch_size,
cutoff_date.isoformat(),
_cutoff_text(cutoff_date),
total_deleted,
type(batch_exc).__name__,
batch_exc,
@ -454,7 +465,7 @@ class SpendLogCleanup:
async def _finish_table(
self,
prisma_client: PrismaClient,
cutoff_date: datetime,
cutoff_date: Cutoff,
table_name: str,
time_column: str,
rows_deleted: int,
@ -541,6 +552,18 @@ class SpendLogCleanup:
deadline=deadline,
)
async def _delete_old_daily_tag_spend_rows(
self, prisma_client: PrismaClient, cutoff_day: str, deadline: float
) -> TableCleanupResult:
return await self._delete_old_rows_batched(
prisma_client,
cutoff_day,
table_name="LiteLLM_DailyTagSpend",
key_columns=("id",),
time_column="date",
deadline=deadline,
)
async def _clean_spend_log_tables(
self, prisma_client: PrismaClient, deadline: float
) -> tuple[TableCleanupResult, ...]:
@ -624,6 +647,18 @@ class SpendLogCleanup:
)
return (health_checks_result,)
async def _clean_daily_tag_spend(
self, prisma_client: PrismaClient, retention_seconds: int, deadline: float
) -> tuple[TableCleanupResult, ...]:
"""
Prune per-day tag spend rows whose ISO day sorts before the horizon day; the horizon day itself is kept.
"""
horizon: Final = datetime.now(timezone.utc) - timedelta(seconds=float(retention_seconds))
cutoff_day: Final = horizon.date().isoformat()
result: Final = await self._delete_old_daily_tag_spend_rows(prisma_client, cutoff_day, deadline)
verbose_proxy_logger.info("Deleted %s expired daily tag spend rows", result.rows_deleted)
return (result,)
@staticmethod
def _run_outcome(results: tuple[TableCleanupResult, ...]) -> RunOutcome:
"""
@ -671,10 +706,14 @@ class SpendLogCleanup:
"maximum_autorouter_session_retention_period"
)
health_check_retention_seconds: Final = self._retention_seconds_for("maximum_health_check_retention_period")
daily_tag_spend_retention_seconds: Final = self._retention_seconds_for(
"maximum_daily_tag_spend_retention_period"
)
if (
not delete_spend_logs
and autorouter_retention_seconds is None
and health_check_retention_seconds is None
and daily_tag_spend_retention_seconds is None
):
SpendLogCleanupMetrics.record_run("skipped_disabled")
return
@ -706,6 +745,7 @@ class SpendLogCleanup:
int(delete_spend_logs and self.retention_seconds is not None)
+ int(autorouter_retention_seconds is not None)
+ int(health_check_retention_seconds is not None)
+ int(daily_tag_spend_retention_seconds is not None)
)
spend_log_results: Final = (
@ -716,8 +756,13 @@ class SpendLogCleanup:
if delete_spend_logs and self.retention_seconds is not None
else ()
)
remaining_groups_after_spend_logs: Final = int(autorouter_retention_seconds is not None) + int(
health_check_retention_seconds is not None
remaining_groups_after_spend_logs: Final = (
int(autorouter_retention_seconds is not None)
+ int(health_check_retention_seconds is not None)
+ int(daily_tag_spend_retention_seconds is not None)
)
remaining_groups_after_sessions: Final = int(health_check_retention_seconds is not None) + int(
daily_tag_spend_retention_seconds is not None
)
session_results: Final = (
await self._clean_session_rollup(
@ -732,13 +777,18 @@ class SpendLogCleanup:
await self._clean_health_checks(
prisma_client,
health_check_retention_seconds,
deadline,
self._group_deadline(deadline, remaining_groups_after_sessions),
)
if health_check_retention_seconds is not None
else ()
)
daily_tag_spend_results: Final = (
await self._clean_daily_tag_spend(prisma_client, daily_tag_spend_retention_seconds, deadline)
if daily_tag_spend_retention_seconds is not None
else ()
)
results: Final = spend_log_results + session_results + health_check_results
results: Final = spend_log_results + session_results + health_check_results + daily_tag_spend_results
outcome: Final = self._run_outcome(results)
SpendLogCleanupMetrics.record_run(outcome)
self._log_run_summary(outcome, results, time.monotonic() - run_started_at)

View file

@ -213,6 +213,8 @@ try:
import orjson
import yaml
from apscheduler.schedulers.asyncio import AsyncIOScheduler
from apscheduler.schedulers.base import STATE_STOPPED
from apscheduler.triggers.base import BaseTrigger
from apscheduler.triggers.interval import IntervalTrigger
except ImportError as e:
raise ImportError(f"Missing dependency {e}. Run `pip install 'litellm[proxy]'`")
@ -5078,6 +5080,20 @@ def _current_general_settings() -> Mapping[str, object]:
return general_settings
_CLEANUP_SCHEDULE_KEYS: Final = (
"maximum_spend_logs_retention_period",
"maximum_autorouter_session_retention_period",
"maximum_health_check_retention_period",
"maximum_daily_tag_spend_retention_period",
"maximum_spend_logs_cleanup_cron",
"maximum_spend_logs_retention_interval",
)
def _cleanup_schedule_of(settings: Mapping[str, object]) -> tuple[object, ...]:
return tuple(settings.get(key) for key in _CLEANUP_SCHEDULE_KEYS)
@lru_cache(maxsize=4096)
def _log_ignored_cost_map_copy(model_id: str, fields: tuple[str, ...]) -> None:
verbose_proxy_logger.warning(
@ -5100,6 +5116,8 @@ class ProxyConfig:
self._last_websearch_interception_config: dict[str, object] | None = None
self._last_hashicorp_vault_config: dict[str, object] | None = None
self._last_cyberark_config: dict[str, object] | None = None # mutable-ok: change-detection cache
self._last_cleanup_schedule_attempt: tuple[object, ...] | None = None
self._cleanup_reschedule_failed: bool = False
self._cyberark_boot_env: dict[str, str | None] | None = None # mutable-ok: deployment env snapshot, set once
self.worker_registry: list[WorkerRegistryEntry] = []
self.config_sync_subscriber: ConfigSyncSubscriber | None = None
@ -7455,69 +7473,67 @@ class ProxyConfig:
if scheduler is None:
return
# Remove existing job if it exists
try:
scheduler.remove_job("spend_log_cleanup_job")
verbose_proxy_logger.info("Removed existing spend log cleanup job")
except Exception:
pass # Job might not exist, which is fine
# Schedule new job if retention period is set (not None)
retention_period: Final = general_settings.get("maximum_spend_logs_retention_period")
autorouter_retention: Final = general_settings.get("maximum_autorouter_session_retention_period")
health_check_retention: Final = general_settings.get("maximum_health_check_retention_period")
if retention_period is not None or autorouter_retention is not None or health_check_retention is not None:
from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import (
SpendLogCleanup,
wants_job: Final = any(
general_settings.get(key) is not None
for key in (
"maximum_spend_logs_retention_period",
"maximum_autorouter_session_retention_period",
"maximum_health_check_retention_period",
"maximum_daily_tag_spend_retention_period",
)
)
if not wants_job:
if scheduler.get_job("spend_log_cleanup_job") is not None:
scheduler.remove_job("spend_log_cleanup_job")
verbose_proxy_logger.info("Removed existing spend log cleanup job")
return
spend_log_cleanup: Final = SpendLogCleanup()
cleanup_cron: Final = general_settings.get("maximum_spend_logs_cleanup_cron")
trigger: Final = self._spend_log_cleanup_trigger()
if trigger is None:
return
from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import (
SpendLogCleanup,
)
if cleanup_cron:
from apscheduler.triggers.cron import CronTrigger
scheduler.add_job(
SpendLogCleanup().cleanup_old_spend_logs,
trigger,
args=[prisma_client],
id="spend_log_cleanup_job",
replace_existing=True,
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
)
verbose_proxy_logger.info("Spend log cleanup rescheduled with trigger: %s", trigger)
try:
cron_trigger: Final = CronTrigger.from_crontab(cleanup_cron)
scheduler.add_job(
spend_log_cleanup.cleanup_old_spend_logs,
cron_trigger,
args=[prisma_client],
id="spend_log_cleanup_job",
replace_existing=True,
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
)
verbose_proxy_logger.info("Spend log cleanup rescheduled with cron: %s", cleanup_cron)
except ValueError:
verbose_proxy_logger.error("Invalid maximum_spend_logs_cleanup_cron value: %s", cleanup_cron)
else:
# Interval-based scheduling (existing behavior)
from litellm.litellm_core_utils.duration_parser import (
duration_in_seconds,
)
def _spend_log_cleanup_trigger(self) -> BaseTrigger | None:
cleanup_cron: Final[object] = general_settings.get("maximum_spend_logs_cleanup_cron")
if cleanup_cron:
from apscheduler.triggers.cron import CronTrigger
retention_interval: Final = general_settings.get("maximum_spend_logs_retention_interval", "1d")
try:
interval_seconds: Final = duration_in_seconds(retention_interval)
# this runs against a started scheduler, which the startup stagger sweep
# cannot reach, so the offset is applied here or the job reconverges across
# replicas the first time an admin edits the retention settings
scheduler.add_job(
spend_log_cleanup.cleanup_old_spend_logs,
stagger_trigger(
job_id="spend_log_cleanup_job",
trigger=IntervalTrigger(seconds=interval_seconds),
period_seconds=interval_seconds,
settings=parse_stagger_settings(general_settings),
),
args=[prisma_client],
id="spend_log_cleanup_job",
replace_existing=True,
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
)
verbose_proxy_logger.info("Spend log cleanup rescheduled with interval: %s", retention_interval)
except ValueError:
verbose_proxy_logger.error("Invalid maximum_spend_logs_retention_interval value")
try:
cron_trigger: Final[BaseTrigger] = CronTrigger.from_crontab(cleanup_cron)
except (ValueError, TypeError, AttributeError):
verbose_proxy_logger.error("Invalid maximum_spend_logs_cleanup_cron value: %s", cleanup_cron)
return None
return cron_trigger
retention_interval: Final[object] = general_settings.get("maximum_spend_logs_retention_interval", "1d")
if not isinstance(retention_interval, str):
verbose_proxy_logger.error("Invalid maximum_spend_logs_retention_interval value: %r", retention_interval)
return None
# this runs against a started scheduler, which the startup stagger sweep
# cannot reach, so the offset is applied here or the job reconverges across
# replicas the first time an admin edits the retention settings
try:
interval_seconds: Final = duration_in_seconds(retention_interval)
return stagger_trigger(
job_id="spend_log_cleanup_job",
trigger=IntervalTrigger(seconds=interval_seconds),
period_seconds=interval_seconds,
settings=parse_stagger_settings(general_settings),
)
except (ValueError, OverflowError):
verbose_proxy_logger.error("Invalid maximum_spend_logs_retention_interval value: %r", retention_interval)
return None
async def _update_general_settings(self, db_general_settings: Mapping[str, SettingsJsonValue] | None) -> None:
global general_settings
@ -7526,32 +7542,28 @@ class ProxyConfig:
if not isinstance(general_settings, SettingsStore):
self.settings.load_yaml(_as_settings_mapping(general_settings))
cache_size_was_db: Final = self.settings.source("user_api_key_cache_max_size") == "db"
previous_retention_values: Final = self._resolved_retention_values()
previous_cleanup_schedule: Final = self._resolved_cleanup_schedule()
previous_pass_through_endpoints: Final = self.settings.get("pass_through_endpoints")
self.settings.apply_db_row("general_settings", db_general_settings)
_bind_general_settings_store(self.settings)
await self._apply_general_settings_side_effects(
db_general_settings,
cache_size_was_db,
previous_retention_values,
previous_cleanup_schedule,
previous_pass_through_endpoints,
)
def _resolved_retention_values(self) -> tuple[SettingsJsonValue | None, ...]:
return tuple(
self.settings.get(key)
for key in (
"maximum_spend_logs_retention_period",
"maximum_autorouter_session_retention_period",
"maximum_health_check_retention_period",
)
)
def _resolved_cleanup_schedule(self) -> tuple[object, ...]:
return _cleanup_schedule_of(self.settings)
def record_cleanup_schedule_attempt(self, settings: Mapping[str, object]) -> None:
self._last_cleanup_schedule_attempt = _cleanup_schedule_of(settings)
async def _apply_general_settings_side_effects(
self,
db_values: Mapping[str, SettingsJsonValue],
cache_size_was_db: bool,
previous_retention_values: tuple[SettingsJsonValue | None, ...],
previous_cleanup_schedule: tuple[object, ...],
previous_pass_through_endpoints: SettingsJsonValue | None,
) -> None:
effects: Final = (
@ -7560,7 +7572,7 @@ class ProxyConfig:
self._apply_boolean_settings,
partial(self._apply_cache_size_setting, cache_size_was_db=cache_size_was_db),
self._apply_store_model_in_db_setting,
partial(self._apply_retention_settings, previous_retention_values=previous_retention_values),
partial(self._apply_retention_settings, previous_cleanup_schedule=previous_cleanup_schedule),
self._apply_ssrf_settings,
)
for effect in effects:
@ -7655,10 +7667,36 @@ class ProxyConfig:
async def _apply_retention_settings(
self,
db_values: Mapping[str, SettingsJsonValue],
previous_retention_values: tuple[SettingsJsonValue | None, ...],
previous_cleanup_schedule: tuple[object, ...],
) -> None:
if previous_retention_values != self._resolved_retention_values():
# while the scheduler is still stopped the startup block owns the first registration
if scheduler is not None and scheduler.state == STATE_STOPPED:
return
schedule: Final = self._resolved_cleanup_schedule()
wants_job: Final = any(value is not None for value in schedule[:4])
has_job: Final = scheduler is not None and scheduler.get_job("spend_log_cleanup_job") is not None
baseline: Final = (
self._last_cleanup_schedule_attempt
if has_job and self._last_cleanup_schedule_attempt is not None
else previous_cleanup_schedule
)
retry_due: Final = (
wants_job
and (not has_job or self._cleanup_reschedule_failed)
and schedule != self._last_cleanup_schedule_attempt
)
if not (baseline != schedule or retry_due or (has_job and not wants_job)):
return
try:
await self._reschedule_spend_log_cleanup_job()
except Exception as exc:
self._cleanup_reschedule_failed = True
verbose_proxy_logger.exception(
"Spend log cleanup could not be rescheduled, will retry on next sync: %s", exc
)
return
self._cleanup_reschedule_failed = False
self._last_cleanup_schedule_attempt = schedule
async def _apply_ssrf_settings(self, db_values: Mapping[str, SettingsJsonValue]) -> None:
_apply_ssrf_general_settings(db_values)
@ -10535,15 +10573,19 @@ class ProxyStartupEvent:
)
### SPEND LOG CLEANUP ###
cleanup_settings: Final = _current_general_settings()
if (
general_settings.get("maximum_spend_logs_retention_period") is not None
or general_settings.get("maximum_autorouter_session_retention_period") is not None
or general_settings.get("maximum_health_check_retention_period") is not None
cleanup_settings.get("maximum_spend_logs_retention_period") is not None
or cleanup_settings.get("maximum_autorouter_session_retention_period") is not None
or cleanup_settings.get("maximum_health_check_retention_period") is not None
or cleanup_settings.get("maximum_daily_tag_spend_retention_period") is not None
):
spend_log_cleanup: Final = SpendLogCleanup()
cleanup_cron: Final = general_settings.get("maximum_spend_logs_cleanup_cron")
cleanup_cron: Final = cleanup_settings.get("maximum_spend_logs_cleanup_cron")
if cleanup_cron:
if cleanup_cron and not isinstance(cleanup_cron, str):
verbose_proxy_logger.error("Invalid maximum_spend_logs_cleanup_cron value: %r", cleanup_cron)
elif isinstance(cleanup_cron, str) and cleanup_cron:
from apscheduler.triggers.cron import CronTrigger
try:
@ -10561,8 +10603,10 @@ class ProxyStartupEvent:
verbose_proxy_logger.error("Invalid maximum_spend_logs_cleanup_cron value: %s", cleanup_cron)
else:
# Interval-based scheduling (existing behavior)
retention_interval: Final = general_settings.get("maximum_spend_logs_retention_interval", "1d")
retention_interval: Final = cleanup_settings.get("maximum_spend_logs_retention_interval", "1d")
try:
if not isinstance(retention_interval, str):
raise ValueError(retention_interval)
interval_seconds: Final = duration_in_seconds(retention_interval)
scheduler.add_job(
spend_log_cleanup.cleanup_old_spend_logs,
@ -10573,8 +10617,11 @@ class ProxyStartupEvent:
replace_existing=True,
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
)
except ValueError:
verbose_proxy_logger.error("Invalid maximum_spend_logs_retention_interval value")
except (ValueError, OverflowError):
verbose_proxy_logger.error(
"Invalid maximum_spend_logs_retention_interval value: %r", retention_interval
)
proxy_config.record_cleanup_schedule_attempt(cleanup_settings)
### CHECK BATCH COST ###
if llm_router is not None and PROXY_BATCH_POLLING_ENABLED:
try:
@ -17909,6 +17956,7 @@ _GENERAL_SETTINGS_CONFIG_LIST_FIELD_TYPES: Final[Mapping[str, str]] = MappingPro
"store_prompts_in_spend_logs": "Boolean",
"maximum_spend_logs_retention_period": "String",
"maximum_health_check_retention_period": "String",
"maximum_daily_tag_spend_retention_period": "String",
"maximum_spend_logs_cleanup_batch_size": "Integer",
"maximum_spend_logs_cleanup_max_batches": "Integer",
"maximum_spend_logs_cleanup_run_budget": "String",

View file

@ -48,6 +48,7 @@ from typing_extensions import NotRequired, ReadOnly, Required, TypedDict
from litellm._logging import verbose_logger
from litellm._uuid import uuid
from litellm.constants import INTERNAL_KWARG_PREFIX
from litellm.types.llms.base import (
BaseLiteLLMOpenAIResponseObject,
CachedTokensDetails,
@ -3937,6 +3938,10 @@ all_litellm_params = [ # rebind-ok: two star imports in litellm/__init__.py re-
]
def is_litellm_owned_kwarg(name: str) -> bool:
return name in all_litellm_params or name.startswith(INTERNAL_KWARG_PREFIX)
class KeyGenerationConfig(TypedDict, total=False):
required_params: list[str] # specify params that must be present in the key generation request

View file

@ -257,7 +257,7 @@ from litellm.types.utils import (
TextCompletionResponse,
TranscriptionResponse,
Usage,
all_litellm_params,
is_litellm_owned_kwarg,
)
_CALL_TYPE_ENUM_MAP: Final[dict] = {ct.value: ct for ct in CallTypes}
@ -4161,26 +4161,8 @@ def _remove_unsupported_params(non_default_params: dict, supported_openai_params
return non_default_params
def filter_out_litellm_params(kwargs: dict) -> dict:
"""
Filter out LiteLLM internal parameters from kwargs dict.
Returns a new dict containing only non-LiteLLM parameters that should be
passed to external provider APIs.
Args:
kwargs: Dictionary that may contain LiteLLM internal parameters
Returns:
Dictionary with LiteLLM internal parameters filtered out
Example:
>>> kwargs = {"query": "test", "shared_session": session_obj, "metadata": {}}
>>> filtered = filter_out_litellm_params(kwargs)
>>> # filtered = {"query": "test"}
"""
return {key: value for key, value in kwargs.items() if key not in all_litellm_params}
def filter_out_litellm_params(kwargs: Mapping[str, object]) -> dict:
return {key: value for key, value in kwargs.items() if not is_litellm_owned_kwarg(key)}
def _provider_supports_vertex_params(custom_llm_provider: str) -> bool:
@ -10152,10 +10134,9 @@ def get_standard_openai_params(params: Mapping[str, object]) -> dict:
def get_non_default_completion_params(kwargs: Mapping[str, object]) -> dict:
openai_params: Final = litellm.OPENAI_CHAT_COMPLETION_PARAMS
default_params: Final = openai_params + all_litellm_params
non_default_params: Final = {
k: v for k, v in kwargs.items() if k not in default_params
} # model-specific params - pass them straight to the model/provider
k: v for k, v in kwargs.items() if k not in openai_params and not is_litellm_owned_kwarg(k)
}
return non_default_params
@ -10203,11 +10184,12 @@ def strip_reasoning_summary_aliases_from_optional_params(
return op, rs_val
def get_non_default_transcription_params(kwargs: dict) -> dict:
def get_non_default_transcription_params(kwargs: Mapping[str, object]) -> dict:
from litellm.constants import OPENAI_TRANSCRIPTION_PARAMS
default_params: Final = OPENAI_TRANSCRIPTION_PARAMS + all_litellm_params
non_default_params: Final = {k: v for k, v in kwargs.items() if k not in default_params}
non_default_params: Final = {
k: v for k, v in kwargs.items() if k not in OPENAI_TRANSCRIPTION_PARAMS and not is_litellm_owned_kwarg(k)
}
return non_default_params

View file

@ -12216,7 +12216,8 @@
"text",
"image"
],
"supports_embedding_image_input": true
"supports_embedding_image_input": true,
"input_cost_per_image_token": 4.7e-07
},
"azure_ai/grok-4": {
"input_cost_per_token": 3e-06,
@ -15804,7 +15805,9 @@
"mode": "embedding",
"output_cost_per_token": 0.0,
"output_vector_size": 1536,
"supports_embedding_image_input": true
"supports_embedding_image_input": true,
"input_cost_per_image_token": 4.7e-07,
"source": "https://cohere.com/pricing"
},
"cohere/parse-v5.0": {
"litellm_provider": "cohere",
@ -41913,14 +41916,14 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v4-pro": {
"cache_read_input_token_cost": 3.828e-08,
"input_cost_per_token": 4.5936e-07,
"cache_read_input_token_cost": 2.9e-08,
"input_cost_per_token": 3.48e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
"max_tokens": 384000,
"mode": "chat",
"output_cost_per_token": 9.1872e-07,
"output_cost_per_token": 6.96e-07,
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
"supports_function_calling": true,
@ -41933,14 +41936,14 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v4.1-flash": {
"cache_read_input_token_cost": 4.2e-09,
"input_cost_per_token": 1.4e-07,
"cache_read_input_token_cost": 1e-09,
"input_cost_per_token": 3.5e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"max_output_tokens": 384000,
"max_tokens": 384000,
"mode": "chat",
"output_cost_per_token": 4.2e-07,
"output_cost_per_token": 2.9e-07,
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
"supports_function_calling": true,
@ -43052,7 +43055,6 @@
"supports_web_search": false
},
"openrouter/openai/gpt-oss-20b": {
"cache_read_input_token_cost": 3e-08,
"input_cost_per_token": 1.8e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 131072,
@ -43322,8 +43324,8 @@
"input_cost_per_token": 2.6e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 65536,
"max_tokens": 65536,
"max_output_tokens": 235929,
"max_tokens": 235929,
"mode": "chat",
"output_cost_per_token": 2.08e-06,
"source": "https://openrouter.ai/api/v1/models",
@ -43603,8 +43605,8 @@
"input_cost_per_token": 9.646e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 204800,
"max_output_tokens": 128000,
"max_tokens": 128000,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 3.0316e-06,
"source": "https://openrouter.ai/api/v1/models",
@ -66825,14 +66827,14 @@
"supports_web_search": false
},
"openrouter/z-ai/glm-5.3-flash": {
"cache_read_input_token_cost": 1e-08,
"input_cost_per_token": 4.5e-08,
"cache_read_input_token_cost": 1.5e-08,
"input_cost_per_token": 4e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1310720,
"max_output_tokens": 943718,
"max_tokens": 943718,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 1.4e-07,
"output_cost_per_token": 5e-07,
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
"supports_function_calling": true,
@ -66865,13 +66867,13 @@
"supports_web_search": false
},
"openrouter/z-ai/glm-5.3": {
"input_cost_per_token": 1.4e-06,
"output_cost_per_token": 4.4e-06,
"cache_read_input_token_cost": 2.6e-07,
"input_cost_per_token": 3.794e-07,
"output_cost_per_token": 1.1924e-06,
"cache_read_input_token_cost": 7.046e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1310720,
"max_output_tokens": 943717,
"max_tokens": 943717,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
@ -67520,8 +67522,8 @@
"input_cost_per_token": 3.2e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 262140,
"max_tokens": 262140,
"max_output_tokens": 81920,
"max_tokens": 81920,
"mode": "chat",
"output_cost_per_token": 3.2e-06,
"source": "https://openrouter.ai/api/v1/models",
@ -68297,8 +68299,8 @@
"input_cost_per_token": 1.5e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 32768,
"max_tokens": 32768,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 6e-07,
"source": "https://openrouter.ai/api/v1/models",
@ -68476,8 +68478,8 @@
"cache_read_input_token_cost": 7e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 16384,
"max_tokens": 16384,
"max_output_tokens": 235929,
"max_tokens": 235929,
"mode": "chat",
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
@ -68637,8 +68639,8 @@
"output_cost_per_token": 3e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 262144,
"max_output_tokens": 32000,
"max_tokens": 32000,
"max_output_tokens": 235929,
"max_tokens": 235929,
"mode": "chat",
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
@ -69502,7 +69504,9 @@
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 1.5e-07,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 128000,
"max_tokens": 128000
},
"vertex_ai/gemini-2.5-flash-native-audio": {
"deprecation_date": "2026-12-13",
@ -69613,42 +69617,72 @@
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 3.5e-06,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 4096,
"max_tokens": 4096
},
"together_ai/meta-llama/Llama-3.2-1B-Instruct": {
"input_cost_per_token": 6e-08,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 6e-08,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 131072,
"max_tokens": 131072
},
"together_ai/meta-llama/Llama-3.2-3B-Instruct": {
"input_cost_per_token": 6e-08,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 6e-08,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 131072,
"max_tokens": 131072
},
"together_ai/Qwen/Qwen2-1.5B-Instruct": {
"input_cost_per_token": 2e-08,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 2e-08,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 32768,
"max_tokens": 32768
},
"together_ai/Qwen/Qwen2.5-14B-Instruct": {
"input_cost_per_token": 8e-07,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 8e-07,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 32768,
"max_tokens": 32768
},
"together_ai/Qwen/Qwen2.5-72B-Instruct": {
"input_cost_per_token": 1.2e-06,
"litellm_provider": "together_ai",
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"source": "https://api.together.ai/v1/models"
"source": "https://api.together.ai/v1/models",
"max_input_tokens": 32768,
"max_tokens": 32768
},
"together_ai/Salesforce/Llama-Rank-V1": {
"input_cost_per_token": 1e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 8192,
"max_tokens": 8192,
"mode": "rerank",
"output_cost_per_token": 0.0,
"source": "https://api.together.xyz/v1/models"
},
"together_ai/meta-llama/Meta-Llama-3.1-8B": {
"input_cost_per_token": 2e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 16384,
"max_tokens": 16384,
"mode": "completion",
"output_cost_per_token": 2e-07,
"source": "https://api.together.xyz/v1/models"
},
"together_ai/together/Tev1-4B-experimental": {
"cache_read_input_token_cost": 4.2e-08,

View file

@ -276,7 +276,7 @@ def provider_wire_environment(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -
@pytest.mark.parametrize("provider", PROVIDERS)
@pytest.mark.parametrize("asynchronous", [False, True])
@pytest.mark.parametrize("stream", [False, True])
async def test_stream_chunk_size_never_reaches_provider_body(
async def test_internal_params_never_reach_provider_body(
monkeypatch: pytest.MonkeyPatch,
provider_wire_environment: None,
provider: str,
@ -289,6 +289,7 @@ async def test_stream_chunk_size_never_reaches_provider_body(
**_request_parameters(provider, wire.url),
"stream": stream,
"stream_chunk_size": 64,
"_litellm_undeclared_sentinel": "internal",
"extra_body": {"custom_provider_key": 1},
"max_tokens": 16,
"timeout": 5,
@ -313,4 +314,5 @@ async def test_stream_chunk_size_never_reaches_provider_body(
keys: Final = keys_at_every_depth(body)
assert "stream_chunk_size" not in keys
assert not INTERNAL_FIELDS.intersection(keys)
assert not frozenset(key for key in keys if key.startswith("_litellm_")), keys
assert _custom_key(body, provider) == 1

View file

@ -0,0 +1,247 @@
import json
import os
import signal
import uuid
from datetime import datetime, timedelta, timezone
from pathlib import Path
from typing import Final
import psutil
import psycopg
import pytest
import yaml
from pydantic import JsonValue, TypeAdapter
from tests.integration._support.client import Gateway, eventually, string_value
from tests.integration._support.database import read_rows
from tests.integration._support.process import OwnedProxy, owned_proxy, owned_proxy_process
CLEANUP_EVERY_MINUTE: Final = "* * * * *"
RETENTION_SETTING: Final = "maximum_daily_tag_spend_retention_period"
_MAPPING: Final = TypeAdapter(dict[str, JsonValue])
_SETTINGS: Final = TypeAdapter(list[dict[str, JsonValue]])
def _day(days_ago: int) -> str:
return (datetime.now(timezone.utc) - timedelta(days=days_ago)).strftime("%Y-%m-%d")
def _seed_daily_tag_spend(tag: str, days: tuple[str, ...]) -> None:
with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection:
for day in days:
connection.execute(
'INSERT INTO "LiteLLM_DailyTagSpend" (id, tag, date, api_key, model, spend, updated_at) '
"VALUES (%s, %s, %s, %s, %s, 1.0, now())",
(uuid.uuid4().hex, tag, day, f"integration-{tag}", "gpt-4o-mini"),
)
def _seed_old_spend_log(request_id: str, days_ago: int) -> None:
with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection:
connection.execute(
'INSERT INTO "LiteLLM_SpendLogs" (request_id, call_type, api_key, spend, "startTime", "endTime") '
"VALUES (%s, 'acompletion', %s, 0, now() - make_interval(days => %s), now() - make_interval(days => %s))",
(request_id, f"integration-{request_id}", str(days_ago), str(days_ago)),
)
def _delete_daily_tag_spend(tag: str) -> None:
with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection:
connection.execute('DELETE FROM "LiteLLM_DailyTagSpend" WHERE tag = %s', (tag,))
def _remaining_days(tag: str) -> tuple[str, ...]:
rows: Final = read_rows('SELECT date FROM "LiteLLM_DailyTagSpend" WHERE tag = %s ORDER BY date', (tag,))
return tuple(str(row["date"]) for row in rows)
def _spend_log_present(request_id: str) -> bool:
return bool(read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id = %s', (request_id,)))
def _stored_retention_setting() -> JsonValue:
rows: Final = read_rows(
'SELECT param_value -> %s AS value FROM "LiteLLM_Config" WHERE param_name = %s',
(RETENTION_SETTING, "general_settings"),
)
return rows[0]["value"] if rows else None
def _store_retention_setting(value: JsonValue) -> None:
with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection:
if value is None:
connection.execute(
'UPDATE "LiteLLM_Config" SET param_value = param_value - %s WHERE param_name = %s',
(RETENTION_SETTING, "general_settings"),
)
return
connection.execute(
'UPDATE "LiteLLM_Config" SET param_value = jsonb_set(param_value, ARRAY[%s], %s::jsonb) '
"WHERE param_name = %s",
(RETENTION_SETTING, json.dumps(value), "general_settings"),
)
def _listening_workers(owned: OwnedProxy) -> tuple[psutil.Process, ...]:
port: Final = owned.gateway.client.base_url.port
return tuple(
child
for child in psutil.Process(owned.process.pid).children(recursive=True)
if any(conn.status == psutil.CONN_LISTEN and conn.laddr.port == port for conn in child.net_connections("inet"))
)
def _listed_retention_value(gateway: Gateway) -> JsonValue:
listed: Final = _SETTINGS.validate_json(
gateway.request("GET", "/config/list", params={"config_type": "general_settings"}).content
)
matching: Final = tuple(entry for entry in listed if entry["field_name"] == RETENTION_SETTING)
return matching[0]["field_value"] if matching else "not listed"
def _completion_id(gateway: Gateway, model: str) -> str:
return string_value(gateway.chat(model, text=f"retention audit {uuid.uuid4().hex}")["id"])
def _cleanup_config(tmp_path: Path, retention: dict[str, JsonValue]) -> Path:
base: Final = _MAPPING.validate_python(yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()))
config: Final = {
**base,
"general_settings": {
**_MAPPING.validate_python(base["general_settings"]),
**retention,
"maximum_spend_logs_cleanup_cron": CLEANUP_EVERY_MINUTE,
"scheduled_job_stagger": {"enabled": False},
},
}
path: Final = tmp_path / "retention.yaml"
path.write_text(yaml.safe_dump(config))
return path
@pytest.mark.timeout(240)
def test_daily_tag_spend_retention_prunes_only_rows_older_than_the_period(gateway: Gateway, tmp_path: Path) -> None:
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
expired, on_the_cutoff, today = _day(200), _day(30), _day(0)
_seed_daily_tag_spend(tag, (expired, on_the_cutoff, today))
try:
config: Final = _cleanup_config(tmp_path, {"maximum_daily_tag_spend_retention_period": "30d"})
with owned_proxy(gateway, tmp_path, {}, config=config):
remaining: Final = eventually(
lambda: _remaining_days(tag),
lambda days: expired not in days,
seconds=150,
)
assert remaining == (on_the_cutoff, today), remaining
finally:
_delete_daily_tag_spend(tag)
@pytest.mark.timeout(240)
def test_config_update_turns_on_daily_tag_spend_cleanup_without_a_restart(gateway: Gateway, tmp_path: Path) -> None:
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
expired, yesterday_of_cutoff, on_the_cutoff, today = _day(200), _day(31), _day(30), _day(0)
_seed_daily_tag_spend(tag, (expired, yesterday_of_cutoff, on_the_cutoff, today))
previously_stored: Final = _stored_retention_setting()
_store_retention_setting(None)
try:
config: Final = _cleanup_config(tmp_path, {})
with owned_proxy(gateway, tmp_path, {}, config=config, workers=2) as owned, owned.scenario() as scenario:
model: Final = scenario.model()
assert _listed_retention_value(owned) is None
owned.post("/config/update", {"general_settings": {RETENTION_SETTING: "30d"}})
assert _listed_retention_value(owned) == "30d"
remaining: Final = eventually(
lambda: _remaining_days(tag),
lambda days: yesterday_of_cutoff not in days,
seconds=150,
)
assert remaining == (on_the_cutoff, today), remaining
assert _completion_id(owned, model).startswith("chatcmpl-")
finally:
_store_retention_setting(previously_stored)
_delete_daily_tag_spend(tag)
@pytest.mark.timeout(240)
def test_unparseable_daily_tag_spend_retention_deletes_nothing_and_keeps_serving(
gateway: Gateway, tmp_path: Path
) -> None:
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
request_id: Final = f"integration-retention-{uuid.uuid4().hex}"
expired: Final = _day(200)
_seed_daily_tag_spend(tag, (expired,))
_seed_old_spend_log(request_id, days_ago=200)
try:
config: Final = _cleanup_config(
tmp_path, {RETENTION_SETTING: "soon", "maximum_spend_logs_retention_period": "30d"}
)
with owned_proxy(gateway, tmp_path, {}, config=config) as owned, owned.scenario() as scenario:
model: Final = scenario.model()
eventually(lambda: _spend_log_present(request_id), lambda present: not present, seconds=150)
assert _remaining_days(tag) == (expired,)
assert _completion_id(owned, model).startswith("chatcmpl-")
finally:
_delete_daily_tag_spend(tag)
@pytest.mark.timeout(240)
def test_daily_tag_spend_keeps_days_the_shorter_spend_log_horizon_already_pruned(
gateway: Gateway, tmp_path: Path
) -> None:
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
request_id: Final = f"integration-retention-{uuid.uuid4().hex}"
expired, inside_tag_horizon = _day(200), _day(60)
_seed_daily_tag_spend(tag, (expired, inside_tag_horizon))
_seed_old_spend_log(request_id, days_ago=60)
try:
config: Final = _cleanup_config(
tmp_path, {RETENTION_SETTING: "90d", "maximum_spend_logs_retention_period": "30d"}
)
with owned_proxy(gateway, tmp_path, {}, config=config):
eventually(lambda: _spend_log_present(request_id), lambda present: not present, seconds=150)
remaining: Final = eventually(lambda: _remaining_days(tag), lambda days: expired not in days, seconds=150)
assert remaining == (inside_tag_horizon,), remaining
finally:
_delete_daily_tag_spend(tag)
@pytest.mark.timeout(240)
def test_daily_tag_spend_cleanup_completes_after_one_of_two_workers_is_killed(gateway: Gateway, tmp_path: Path) -> None:
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
expired, today = _day(200), _day(0)
_seed_daily_tag_spend(tag, (expired, today))
try:
config: Final = _cleanup_config(tmp_path, {RETENTION_SETTING: "30d"})
with owned_proxy_process(gateway, tmp_path, {}, config=config, workers=2) as owned:
with owned.gateway.scenario() as scenario:
model: Final = scenario.model()
workers: Final = eventually(
lambda: _listening_workers(owned), lambda found: len(found) == 2, seconds=30
)
workers[0].send_signal(signal.SIGKILL)
eventually(lambda: workers[0].is_running(), lambda alive: not alive, seconds=10)
ids: Final = tuple(_completion_id(owned.gateway, model) for _ in range(6))
assert len(set(ids)) == 6 and all(identity.startswith("chatcmpl-") for identity in ids), ids
remaining: Final = eventually(
lambda: _remaining_days(tag), lambda days: expired not in days, seconds=150
)
assert remaining == (today,), remaining
finally:
_delete_daily_tag_spend(tag)
@pytest.mark.timeout(240)
def test_daily_tag_spend_is_kept_forever_when_its_retention_is_unset(gateway: Gateway, tmp_path: Path) -> None:
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
request_id: Final = f"integration-retention-{uuid.uuid4().hex}"
expired: Final = _day(200)
_seed_daily_tag_spend(tag, (expired,))
_seed_old_spend_log(request_id, days_ago=200)
try:
config: Final = _cleanup_config(tmp_path, {"maximum_spend_logs_retention_period": "30d"})
with owned_proxy(gateway, tmp_path, {}, config=config):
eventually(lambda: _spend_log_present(request_id), lambda present: not present, seconds=150)
assert _remaining_days(tag) == (expired,)
finally:
_delete_daily_tag_spend(tag)

View file

@ -133,46 +133,6 @@ def test_get_llm_provider_azure_o1():
assert model == "o1-mini"
def test_default_api_base():
from litellm.litellm_core_utils.get_llm_provider_logic import (
_get_openai_compatible_provider_info,
)
from litellm.types.utils import LlmProviders
# Patch environment variable to remove API base if it's set
with patch.dict(os.environ, {}, clear=True):
for provider in litellm.openai_compatible_providers:
# Get the API base for the given provider
if provider == "github_copilot":
continue
# Skip chatgpt as it requires OAuth authentication
if provider == "chatgpt":
continue
# Skip ragflow as it requires specific model format: ragflow/chat/{id}/{model} or ragflow/agent/{id}/{model}
if provider == "ragflow":
continue
_, _, _, api_base = _get_openai_compatible_provider_info(
model=f"{provider}/*", api_base=None, api_key=None, dynamic_api_key=None
)
if api_base is None:
continue
for other_provider in LlmProviders:
if other_provider.value != provider and provider != "{}_chat".format(
other_provider.value
):
if provider == "codestral" and other_provider.value == "mistral":
continue
elif provider == "github" and other_provider.value == "azure":
continue
elif (
provider in ("qwencloud", "qwen_ai_platform")
and other_provider.value == "dashscope"
):
continue
assert other_provider.value not in api_base.replace("/openai", "")
def test_hosted_vllm_default_api_key():
from litellm.litellm_core_utils.get_llm_provider_logic import (
_get_openai_compatible_provider_info,

View file

@ -79,6 +79,7 @@ _PREVIOUSLY_DB_WINS: Final[tuple[str, ...]] = (
"maximum_spend_logs_retention_period",
"maximum_autorouter_session_retention_period",
"maximum_health_check_retention_period",
"maximum_daily_tag_spend_retention_period",
"maximum_spend_logs_cleanup_batch_size",
"maximum_spend_logs_cleanup_max_batches",
"maximum_spend_logs_cleanup_run_budget",

View file

@ -3862,6 +3862,7 @@ async def test_ProxyConfig__reschedule_spend_log_cleanup_job_health_check_retent
async def test_ProxyConfig__update_general_settings_updates_health_check_retention(monkeypatch):
settings = {}
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", settings)
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", MagicMock(**{"get_job.return_value": None}))
pc = ProxyConfig()
reschedule = AsyncMock()
monkeypatch.setattr(pc, "_reschedule_spend_log_cleanup_job", reschedule)
@ -3872,6 +3873,329 @@ async def test_ProxyConfig__update_general_settings_updates_health_check_retenti
reschedule.assert_awaited_once()
def _paused_scheduler(monkeypatch):
from apscheduler.schedulers.asyncio import AsyncIOScheduler
real_scheduler = AsyncIOScheduler()
real_scheduler.start(paused=True)
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
return real_scheduler
def _scheduler_whose_first_add_job_raises(monkeypatch):
from apscheduler.schedulers.asyncio import AsyncIOScheduler
class FirstAddJobRaises(AsyncIOScheduler):
raised = False
def add_job(self, *args, **kwargs):
if not self.raised:
self.raised = True
raise RuntimeError("scheduler busy")
return super().add_job(*args, **kwargs)
real_scheduler = FirstAddJobRaises()
real_scheduler.start(paused=True)
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
return real_scheduler
@pytest.mark.asyncio
async def test_ProxyConfig__reschedule_spend_log_cleanup_job_daily_tag_spend_retention(monkeypatch):
real_scheduler = _paused_scheduler(monkeypatch)
monkeypatch.setattr(
"litellm.proxy.proxy_server.general_settings",
{"maximum_daily_tag_spend_retention_period": "90d"},
)
pc = ProxyConfig()
try:
await pc._reschedule_spend_log_cleanup_job()
job = real_scheduler.get_job("spend_log_cleanup_job")
assert job is not None, "daily tag spend retention alone did not schedule the cleanup job"
assert job.func.__name__ == "cleanup_old_spend_logs"
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_updates_daily_tag_spend_retention(monkeypatch):
real_scheduler = _paused_scheduler(monkeypatch)
pc = ProxyConfig()
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
try:
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
from litellm.proxy import proxy_server
assert proxy_server.general_settings["maximum_daily_tag_spend_retention_period"] == "90d"
assert real_scheduler.get_job("spend_log_cleanup_job") is not None, "runtime retention did not schedule cleanup"
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_schedules_cleanup_when_db_row_was_already_applied(monkeypatch):
"""A config reload applies the db row to the store before the side effects run, so the
before/after snapshot is equal; the job must still be scheduled when none is running."""
real_scheduler = _paused_scheduler(monkeypatch)
pc = ProxyConfig()
pc.settings.apply_db_row("general_settings", {"maximum_daily_tag_spend_retention_period": "90d"})
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
try:
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
assert real_scheduler.get_job("spend_log_cleanup_job") is not None, "DB-only retention never scheduled cleanup"
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_retries_a_failed_schedule_once_per_settings_value(
monkeypatch, caplog
):
"""An unparseable cron leaves no job behind; reloads must not retry it every tick, only when the
cron or a retention value changes."""
real_scheduler = _paused_scheduler(monkeypatch)
pc = ProxyConfig()
bad_cron = {"maximum_daily_tag_spend_retention_period": "90d", "maximum_spend_logs_cleanup_cron": "not a cron"}
pc.settings.apply_db_row("general_settings", bad_cron)
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
try:
with caplog.at_level(logging.ERROR, logger="LiteLLM Proxy"):
for _ in range(3):
await pc._update_general_settings(bad_cron)
assert real_scheduler.get_job("spend_log_cleanup_job") is None
cron_errors = [r for r in caplog.records if "maximum_spend_logs_cleanup_cron" in r.getMessage()]
assert len(cron_errors) == 1, f"invalid cron was retried on every reload: {len(cron_errors)} error lines"
await pc._update_general_settings({**bad_cron, "maximum_spend_logs_cleanup_cron": "* * * * *"})
job = real_scheduler.get_job("spend_log_cleanup_job")
assert job is not None, "a corrected cron did not schedule cleanup"
assert "minute='*'" in str(job.trigger)
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_retries_a_schedule_that_raised(monkeypatch):
"""A transient add_job failure must not be remembered as a completed attempt; the next
reload with the same settings tries again."""
real_scheduler = _scheduler_whose_first_add_job_raises(monkeypatch)
pc = ProxyConfig()
retention = {"maximum_daily_tag_spend_retention_period": "90d"}
pc.settings.apply_db_row("general_settings", retention)
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
try:
await pc._update_general_settings(retention)
assert real_scheduler.get_job("spend_log_cleanup_job") is None
await pc._update_general_settings(retention)
assert real_scheduler.get_job("spend_log_cleanup_job") is not None, "raised add_job was not retried"
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_retries_a_failed_replacement_of_the_live_job(monkeypatch):
"""A cron change whose add_job raised keeps the old job running, so the next reload with the
same settings must try the replacement again instead of leaving the new cron unapplied."""
real_scheduler = _scheduler_whose_first_add_job_raises(monkeypatch)
pc = ProxyConfig()
pc.settings.load_yaml({"maximum_daily_tag_spend_retention_period": "90d"})
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
real_scheduler.raised = True
await pc._reschedule_spend_log_cleanup_job()
real_scheduler.raised = False
try:
new_cron = {"maximum_spend_logs_cleanup_cron": "0 3 * * *"}
await pc._update_general_settings(new_cron)
assert "hour='3'" not in str(real_scheduler.get_job("spend_log_cleanup_job").trigger), "old job was lost"
await pc._update_general_settings(new_cron)
assert "hour='3'" in str(real_scheduler.get_job("spend_log_cleanup_job").trigger), (
"failed replacement was not retried on the next sync"
)
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_leaves_a_changed_db_schedule_to_startup_while_scheduler_is_stopped(
monkeypatch,
):
"""The first DB sync runs before the scheduler starts and usually differs from the yaml; it
must still leave registration to the startup block instead of adding a job it will replace."""
from apscheduler.schedulers.asyncio import AsyncIOScheduler
real_scheduler = AsyncIOScheduler()
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
pc = ProxyConfig()
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
assert real_scheduler.get_jobs() == [], "DB sync registered the cleanup job before the scheduler started"
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_leaves_first_registration_to_startup_while_scheduler_is_stopped(
monkeypatch,
):
"""The DB sync that runs before the scheduler starts must not register the cleanup job; the
startup block does, once, so the cross-replica stagger it applies to pending jobs survives."""
from apscheduler.schedulers.asyncio import AsyncIOScheduler
real_scheduler = AsyncIOScheduler()
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
pc = ProxyConfig()
pc.settings.load_yaml({"maximum_daily_tag_spend_retention_period": "90d"})
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
await pc._update_general_settings({"unrelated_key": "value"})
assert real_scheduler.get_jobs() == [], "DB sync registered the cleanup job before the scheduler started"
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_runtime_interval_job_carries_the_stagger_offset(monkeypatch):
"""Once the scheduler is running the sync owns registration and the job it adds is staggered."""
from apscheduler.schedulers.asyncio import AsyncIOScheduler
from litellm.proxy.common_utils.scheduled_job_stagger import _OffsetTrigger
real_scheduler = AsyncIOScheduler()
real_scheduler.start(paused=True)
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
pc = ProxyConfig()
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
try:
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
jobs = real_scheduler.get_jobs()
assert [job.id for job in jobs] == ["spend_log_cleanup_job"]
assert isinstance(jobs[0].trigger, _OffsetTrigger), repr(jobs[0].trigger)
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
@pytest.mark.parametrize(
"bad_schedule",
[
{"maximum_spend_logs_cleanup_cron": "not a cron"},
{"maximum_spend_logs_cleanup_cron": "0 0 * * * *"},
{"maximum_spend_logs_retention_interval": "soon"},
{"maximum_spend_logs_retention_interval": 86400},
],
)
async def test_ProxyConfig__update_general_settings_keeps_the_live_cleanup_job_when_the_new_schedule_is_invalid(
monkeypatch, bad_schedule
):
"""A schedule edit that does not parse must leave the old cleanup job running and must not
stop the rest of the general settings sync."""
from apscheduler.schedulers.asyncio import AsyncIOScheduler
real_scheduler = AsyncIOScheduler()
real_scheduler.start(paused=True)
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
ssrf_sync = MagicMock()
monkeypatch.setattr("litellm.proxy.proxy_server._apply_ssrf_general_settings", ssrf_sync)
pc = ProxyConfig()
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
try:
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
old_trigger = real_scheduler.get_job("spend_log_cleanup_job").trigger
ssrf_sync.reset_mock()
for _ in range(2):
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d", **bad_schedule})
live_job = real_scheduler.get_job("spend_log_cleanup_job")
assert live_job is not None, "invalid schedule removed the cleanup job"
assert live_job.trigger is old_trigger
assert ssrf_sync.call_count == 2, "schedule error blocked the rest of the settings sync"
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_logs_an_overflowing_interval_once(monkeypatch, caplog):
"""An interval that parses but overflows the trigger must keep the live job and log one
error, not a traceback on every sync."""
from apscheduler.schedulers.asyncio import AsyncIOScheduler
real_scheduler = AsyncIOScheduler()
real_scheduler.start(paused=True)
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
pc = ProxyConfig()
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
try:
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
old_trigger = real_scheduler.get_job("spend_log_cleanup_job").trigger
overflowing = {
"maximum_daily_tag_spend_retention_period": "90d",
"maximum_spend_logs_retention_interval": "99999999999d",
}
with caplog.at_level(logging.ERROR, logger="LiteLLM Proxy"):
for _ in range(5):
await pc._update_general_settings(overflowing)
errors = [record for record in caplog.records if record.levelno >= logging.ERROR]
assert len(errors) == 1, [record.getMessage() for record in errors]
assert real_scheduler.get_job("spend_log_cleanup_job").trigger is old_trigger
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_reschedules_when_only_the_cron_changes(monkeypatch):
real_scheduler = _paused_scheduler(monkeypatch)
pc = ProxyConfig()
pc.settings.load_yaml({"maximum_daily_tag_spend_retention_period": "90d"})
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
await pc._reschedule_spend_log_cleanup_job()
try:
interval_job = real_scheduler.get_job("spend_log_cleanup_job")
assert interval_job is not None and "hour='3'" not in str(interval_job.trigger)
await pc._update_general_settings({"maximum_spend_logs_cleanup_cron": "0 3 * * *"})
cron_job = real_scheduler.get_job("spend_log_cleanup_job")
assert "hour='3'" in str(cron_job.trigger), "cron-only change did not reschedule"
await pc._update_general_settings({"maximum_spend_logs_cleanup_cron": "0 3 * * *"})
assert real_scheduler.get_job("spend_log_cleanup_job") is cron_job, "unchanged cron replaced the job"
finally:
real_scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_ProxyConfig__update_general_settings_reschedules_a_cron_edit_the_reload_path_already_applied(
monkeypatch,
):
"""The periodic reload applies the DB row through _update_config_from_db before
_update_general_settings snapshots the previous schedule, so a cron edited in the DB must
still replace the live job's trigger."""
from apscheduler.schedulers.asyncio import AsyncIOScheduler
real_scheduler = AsyncIOScheduler()
real_scheduler.start(paused=True)
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
pc = ProxyConfig()
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
try:
first_row = {"maximum_daily_tag_spend_retention_period": "90d", "maximum_spend_logs_cleanup_cron": "0 3 * * *"}
pc.settings.apply_db_row("general_settings", first_row)
await pc._update_general_settings(first_row)
assert "hour='3'" in str(real_scheduler.get_job("spend_log_cleanup_job").trigger)
edited_row = {**first_row, "maximum_spend_logs_cleanup_cron": "0 5 * * *"}
pc.settings.apply_db_row("general_settings", edited_row)
await pc._update_general_settings(edited_row)
assert "hour='5'" in str(real_scheduler.get_job("spend_log_cleanup_job").trigger), "DB cron edit was ignored"
pc.settings.apply_db_row("general_settings", edited_row)
await pc._update_general_settings(edited_row)
assert "hour='5'" in str(real_scheduler.get_job("spend_log_cleanup_job").trigger)
finally:
real_scheduler.shutdown(wait=False)
# ---------------------------------------------------------------------------
# ProxyConfig._update_general_settings
# ---------------------------------------------------------------------------
@ -4003,6 +4327,7 @@ async def test_ProxyConfig__update_general_settings_skips_redundant_retention_re
pc = ProxyConfig()
reschedule: Final = AsyncMock()
monkeypatch.setattr(proxy_server, "general_settings", {})
monkeypatch.setattr(proxy_server, "scheduler", MagicMock())
monkeypatch.setattr(pc, "_reschedule_spend_log_cleanup_job", reschedule)
await pc._update_general_settings({"maximum_health_check_retention_period": "30d"})
@ -4021,6 +4346,7 @@ async def test_ProxyConfig__update_general_settings_reschedules_after_retention_
pc = ProxyConfig()
reschedule: Final = AsyncMock()
monkeypatch.setattr(proxy_server, "general_settings", {})
monkeypatch.setattr(proxy_server, "scheduler", MagicMock(**{"get_job.return_value": None}))
monkeypatch.setattr(pc, "_reschedule_spend_log_cleanup_job", reschedule)
await pc._update_general_settings({"maximum_health_check_retention_period": "30d"})
@ -4052,7 +4378,7 @@ async def test_ProxyConfig__update_general_settings_dispatches_every_side_effect
if name == "_apply_cache_size_setting":
handler.assert_awaited_once_with({}, cache_size_was_db=False)
elif name == "_apply_retention_settings":
handler.assert_awaited_once_with({}, previous_retention_values=())
handler.assert_awaited_once_with({}, previous_cleanup_schedule=())
elif name == "_apply_pass_through_settings":
handler.assert_awaited_once_with({}, previous_endpoints=None)
else:

View file

@ -935,6 +935,89 @@ async def test_periodic_reload_job_scheduled_without_store_model_in_db(monkeypat
scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_initialize_scheduled_jobs_registers_cleanup_when_retention_lives_only_in_the_db(monkeypatch):
"""With no config file, the startup DB sync rebinds general_settings to a store holding the
retention period; the cleanup job must be registered from that live value, not the stale
empty dict the caller passed in."""
monkeypatch.delenv("DISABLE_PRISMA_SCHEMA_UPDATE", raising=False)
monkeypatch.delenv("STORE_MODEL_IN_DB", raising=False)
from apscheduler.schedulers.asyncio import AsyncIOScheduler
from litellm.proxy.proxy_server import ProxyStartupEvent
from litellm.proxy.utils import ProxyLogging
mock_prisma_client = MagicMock()
mock_prisma_client.db.litellm_config.find_first = AsyncMock(return_value=None)
mock_proxy_logging = MagicMock(spec=ProxyLogging)
mock_proxy_logging.slack_alerting_instance = MagicMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_config = _mock_scheduled_proxy_config()
db_settings = proxy_server_module.ProxyConfig().settings
db_settings.apply_db_row("general_settings", {"maximum_daily_tag_spend_retention_period": "30d"})
async def sync_from_db(*args: object, **kwargs: object) -> None:
proxy_server_module._bind_general_settings_store(db_settings)
mock_proxy_config.add_deployment.side_effect = sync_from_db
scheduler = AsyncIOScheduler()
try:
with (
patch("litellm.proxy.proxy_server.proxy_config", mock_proxy_config),
patch("litellm.proxy.proxy_server.store_model_in_db", True),
patch("litellm.proxy.proxy_server.general_settings", {}),
patch("litellm.proxy.proxy_server.AsyncIOScheduler", return_value=scheduler),
):
await ProxyStartupEvent.initialize_scheduled_background_jobs(
general_settings={},
prisma_client=mock_prisma_client,
proxy_budget_rescheduler_min_time=1,
proxy_budget_rescheduler_max_time=2,
proxy_batch_write_at=5,
proxy_logging_obj=mock_proxy_logging,
)
assert scheduler.get_job("spend_log_cleanup_job") is not None, "DB-only retention was not scheduled at boot"
finally:
scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_initialize_scheduled_jobs_does_not_fall_back_to_the_interval_for_a_non_string_cron(monkeypatch):
"""A truthy non-string cron is invalid, so startup must log it and register no cleanup job
rather than silently pruning on the default interval the admin never configured."""
monkeypatch.delenv("DISABLE_PRISMA_SCHEMA_UPDATE", raising=False)
monkeypatch.delenv("STORE_MODEL_IN_DB", raising=False)
from apscheduler.schedulers.asyncio import AsyncIOScheduler
from litellm.proxy.proxy_server import ProxyStartupEvent
from litellm.proxy.utils import ProxyLogging
mock_prisma_client = MagicMock()
mock_proxy_logging = MagicMock(spec=ProxyLogging)
mock_proxy_logging.slack_alerting_instance = MagicMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
settings = {"maximum_daily_tag_spend_retention_period": "30d", "maximum_spend_logs_cleanup_cron": 5}
scheduler = AsyncIOScheduler()
try:
with (
patch("litellm.proxy.proxy_server.proxy_config", _mock_scheduled_proxy_config()),
patch("litellm.proxy.proxy_server.store_model_in_db", False),
patch("litellm.proxy.proxy_server.general_settings", settings),
patch("litellm.proxy.proxy_server.AsyncIOScheduler", return_value=scheduler),
):
await ProxyStartupEvent.initialize_scheduled_background_jobs(
general_settings=settings,
prisma_client=mock_prisma_client,
proxy_budget_rescheduler_min_time=1,
proxy_budget_rescheduler_max_time=2,
proxy_batch_write_at=5,
proxy_logging_obj=mock_proxy_logging,
)
assert scheduler.get_job("spend_log_cleanup_job") is None, "invalid cron fell back to the interval"
finally:
scheduler.shutdown(wait=False)
@pytest.mark.asyncio
async def test_initialize_scheduled_jobs_uses_configured_config_reload_interval(monkeypatch):
"""

View file

@ -827,6 +827,29 @@ async def test_health_check_retention_alone_cleans_only_the_health_check_table()
assert abs((cutoff_date - expected_cutoff).total_seconds()) < 1
@pytest.mark.asyncio
async def test_daily_tag_spend_retention_alone_prunes_only_that_table_by_calendar_day():
client = _mock_prisma_for_retention([0])
cleaner = SpendLogCleanup(general_settings={"maximum_daily_tag_spend_retention_period": "90d"})
cleaner.pod_lock_manager = None
await cleaner.cleanup_old_spend_logs(client)
tables = [call[0][0] for call in client.db.execute_raw.call_args_list]
assert len(tables) == 1
assert '"LiteLLM_DailyTagSpend"' in tables[0]
cutoff_day = client.db.execute_raw.call_args[0][1]
assert cutoff_day == (datetime.now(timezone.utc) - timedelta(days=90)).date().isoformat()
@pytest.mark.asyncio
async def test_spend_logs_retention_alone_keeps_daily_tag_spend_forever():
client = _mock_prisma_for_retention([0, 0])
cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cleaner.pod_lock_manager = None
await cleaner.cleanup_old_spend_logs(client)
tables = [call[0][0] for call in client.db.execute_raw.call_args_list]
assert not any('"LiteLLM_DailyTagSpend"' in sql for sql in tables)
@pytest.mark.asyncio
async def test_each_retention_key_cuts_off_at_its_own_horizon():
client = _mock_prisma_for_retention([0, 0, 0, 0, 0])

View file

@ -58,6 +58,26 @@ def test_image_edit_forwards_provider_params_and_extra_body():
assert response.data
def test_image_edit_keeps_an_internal_prefixed_kwarg_out_of_the_provider_request():
captured = {}
client = HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(_capture_image_edit_request(captured))))
litellm.image_edit(
model="openai/gpt-image-1",
image=PNG_BYTES,
prompt="add a hat",
api_key="sk-test",
api_base="https://edit.example/v1",
client=client,
seed=42,
_litellm_undeclared_sentinel="internal",
)
fields = _multipart_text_fields(captured["content_type"], captured["body"])
assert "_litellm_undeclared_sentinel" not in fields
assert fields["seed"] == "42"
def test_image_edit_extra_body_takes_precedence_over_kwargs():
captured = {}
client = HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(_capture_image_edit_request(captured))))

View file

@ -0,0 +1,29 @@
import json
from typing import Final
import httpx
import respx
import litellm
def test_image_generation_keeps_an_internal_prefixed_kwarg_out_of_the_provider_request(
respx_mock: respx.MockRouter,
) -> None:
api_base: Final = "http://localhost:12346/v1"
mock_route: Final = respx_mock.post(url__regex=rf"{api_base}/images/generations.*").mock(
return_value=httpx.Response(status_code=200, json={"created": 1712697600, "data": [{"b64_json": "aW1n"}]})
)
litellm.image_generation(
model="openai/gpt-image-1",
prompt="a red circle",
api_base=api_base,
api_key="fake_openai_api_key",
_litellm_undeclared_sentinel="internal",
)
assert mock_route.called
sent: Final = json.loads(respx_mock.calls[0].request.content)
assert "_litellm_undeclared_sentinel" not in sent, sent
assert sent["prompt"] == "a red circle"

View file

@ -84,6 +84,29 @@ class TestBedrockFilesTransformation:
"max_tokens" in model_input
), f"Record {i+1} should have max_tokens"
def test_batch_keeps_an_internal_prefixed_key_out_of_the_bedrock_model_input(self):
from litellm.llms.bedrock.files.transformation import BedrockFilesConfig
result: Final = BedrockFilesConfig()._transform_openai_jsonl_content_to_bedrock_jsonl_content(
[
{
"custom_id": "internal-key-1",
"method": "POST",
"url": "/v1/chat/completions",
"body": {
"model": "anthropic.claude-3-5-sonnet-20240620-v1:0",
"messages": [{"role": "user", "content": "hi"}],
"max_tokens": 10,
"_litellm_undeclared_sentinel": "internal",
},
}
]
)
model_input: Final = json.dumps(result[0]["modelInput"])
assert "_litellm_undeclared_sentinel" not in model_input, model_input
assert result[0]["modelInput"]["max_tokens"] == 10
def test_nova_text_only_uses_converse_format(self):
"""
Test that Nova models produce Converse API format in batch modelInput.

View file

@ -1,5 +1,11 @@
import pytest
import json
from typing import Final
import httpx
import pytest
import respx
import litellm
from litellm.llms.elevenlabs.text_to_speech.transformation import (
ElevenLabsTextToSpeechConfig,
)
@ -16,10 +22,7 @@ def test_should_encode_elevenlabs_voice_id_path_segment():
},
)
assert (
url
== "https://api.elevenlabs.io/v1/text-to-speech/voice%2F..%2F..%2Fmodels%3Fx%3D1%23frag"
)
assert url == "https://api.elevenlabs.io/v1/text-to-speech/voice%2F..%2F..%2Fmodels%3Fx%3D1%23frag"
def test_should_reject_dot_segment_elevenlabs_voice_id():
@ -31,3 +34,24 @@ def test_should_reject_dot_segment_elevenlabs_voice_id():
api_base="https://api.elevenlabs.io",
litellm_params={config.ELEVENLABS_VOICE_ID_KEY: ".."},
)
def test_speech_keeps_an_internal_prefixed_kwarg_out_of_the_elevenlabs_request(respx_mock: respx.MockRouter) -> None:
api_base: Final = "http://localhost:12346"
mock_route: Final = respx_mock.post(url__regex=rf"{api_base}/v1/text-to-speech/.*").mock(
return_value=httpx.Response(status_code=200, content=b"audio", headers={"content-type": "audio/mpeg"})
)
litellm.speech(
model="elevenlabs/eleven_multilingual_v2",
input="hi",
voice="21m00Tcm4TlvDq8ikWAM",
api_base=api_base,
api_key="fake_elevenlabs_api_key",
_litellm_undeclared_sentinel="internal",
)
assert mock_route.called
sent: Final = json.loads(respx_mock.calls[0].request.content)
assert "_litellm_undeclared_sentinel" not in sent, sent
assert sent["text"] == "hi"

View file

@ -395,6 +395,34 @@ def test_completion_strips_eager_input_streaming_before_openai(respx_mock: respx
assert sent_tool["function"]["name"] == "write_file"
def test_embedding_keeps_an_internal_prefixed_kwarg_out_of_the_provider_request(respx_mock: respx.MockRouter) -> None:
api_base: Final = "http://localhost:12346/v1"
mock_route: Final = respx_mock.post(url__regex=rf"{api_base}/embeddings.*").mock(
return_value=httpx.Response(
status_code=200,
json={
"object": "list",
"data": [{"object": "embedding", "index": 0, "embedding": [0.1, 0.2]}],
"model": "text-embedding-3-small",
"usage": {"prompt_tokens": 1, "total_tokens": 1},
},
)
)
litellm.embedding(
model="openai/text-embedding-3-small",
input="hi",
api_base=api_base,
api_key="fake_openai_api_key",
_litellm_undeclared_sentinel="internal",
)
assert mock_route.called
sent: Final = json.loads(respx_mock.calls[0].request.content)
assert "_litellm_undeclared_sentinel" not in sent, sent
assert sent["model"] == "text-embedding-3-small"
def test_custom_provider_with_extra_headers():
with patch.object(

View file

@ -262,7 +262,7 @@ OWNED_NAMES: Final = (
*PRICING_NAMES,
)
Classifier: TypeAlias = Callable[[dict[str, object]], dict[str, object]] # mutable-ok: classifiers use dict
Classifier: TypeAlias = Callable[[Mapping[str, object]], Mapping[str, object]]
CLASSIFIERS: Final[Mapping[str, Classifier]] = MappingProxyType(
{ # pyright: ignore[reportUnknownArgumentType] # untyped legacy classifiers
@ -279,18 +279,31 @@ def test_owned_name_is_kept_out_of_provider_params(name: str, classifier_name: s
provider_value: Final = object()
classify: Final = CLASSIFIERS[classifier_name]
result: Final = classify({name: object(), PROVIDER_KNOB: provider_value}) # mutable-ok: classifiers take a dict
result: Final = classify(MappingProxyType({name: object(), PROVIDER_KNOB: provider_value}))
assert result == MappingProxyType({PROVIDER_KNOB: provider_value})
assert result[PROVIDER_KNOB] is provider_value
def test_a_name_no_object_declares_reaches_the_provider() -> None:
result: Final = CLASSIFIERS["completion"]({PROVIDER_KNOB: 1}) # mutable-ok: classifier input type
result: Final = CLASSIFIERS["completion"](MappingProxyType({PROVIDER_KNOB: 1}))
assert result == MappingProxyType({PROVIDER_KNOB: 1})
@pytest.mark.parametrize("classifier_name", CLASSIFIERS)
def test_an_undeclared_internal_prefixed_name_is_kept_out_of_provider_params(classifier_name: str) -> None:
undeclared: Final = "_litellm_never_declared_anywhere"
lookalike: Final = "provider_litellm_knob"
assert undeclared not in all_litellm_params
result: Final = CLASSIFIERS[classifier_name](
MappingProxyType({undeclared: object(), PROVIDER_KNOB: 1, lookalike: 2})
)
assert result == MappingProxyType({PROVIDER_KNOB: 1, lookalike: 2})
def _cache_key_for_model_group(cache: Cache, model_group: str, options: CachingOptions) -> str:
return cache.get_cache_key( # pyright: ignore[reportUnknownMemberType] # untyped legacy key builder
model=model_group,
@ -421,9 +434,7 @@ CARRIED_PARAMS: Final = tuple(
def test_every_param_get_litellm_params_carries_is_kept_out_of_provider_params(name: str) -> None:
provider_value: Final = object()
result: Final = CLASSIFIERS["completion"](
{name: object(), PROVIDER_KNOB: provider_value} # mutable-ok: classifier input type
)
result: Final = CLASSIFIERS["completion"](MappingProxyType({name: object(), PROVIDER_KNOB: provider_value}))
assert result == MappingProxyType({PROVIDER_KNOB: provider_value})

View file

@ -28325,6 +28325,11 @@ export interface components {
* @description Maximum retention period for auto-router benchmark session rollup rows (e.g., '365d'). Rows whose last turn is older than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rollup rows are never deleted.
*/
maximum_autorouter_session_retention_period?: string | null;
/**
* Maximum Daily Tag Spend Retention Period
* @description Maximum retention period for per-day tag spend aggregate rows (e.g., '90d'). Rows whose day is older than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rows are never deleted. Only historical tag usage analytics are affected; tag budgets read the lifetime counter.
*/
maximum_daily_tag_spend_retention_period?: string | null;
/**
* Maximum Health Check Retention Period
* @description Maximum retention period for health-check rows (e.g., '30d'). Rows whose checked_at is older than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rows are never deleted. Set this well above health_check_interval because /health and the UI read the latest row per model.