mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge remote-tracking branch 'origin/main' into litellm_integration_messages_folder
This commit is contained in:
commit
2a7d223318
27 changed files with 1211 additions and 276 deletions
2
.github/pull_request_template.md
vendored
2
.github/pull_request_template.md
vendored
|
|
@ -54,7 +54,7 @@ After: the same request comes back with real token counts, so the dashboard show
|
|||
|
||||
## Affected release
|
||||
|
||||
<!-- Only for a fix to a regression in a released or rc version (perf, memory, crash, or behavior): name the version it regressed in, e.g. "regression in v1.100.0" or "since v1.101.0-rc.1", and add the `backport-stable` label so the fix is cherry-picked onto the rc line before the stable is tagged. Drop the section otherwise -->
|
||||
<!-- Only for a fix to a regression in a released or rc version (perf, memory, crash, or behavior): name the version it regressed in, e.g. "regression in v1.100.0" or "since v1.101.0-rc.1". Add the `backport-stable` label only when the regression is a P0, meaning its Linear ticket is Urgent (a security hole however narrow, data loss, or a crash or outage for every user on that version), because every labeled PR must be cherry-picked onto the baking rc line before the stable can be tagged; every other regression fix ships in the next rc unlabeled. Drop the section otherwise -->
|
||||
|
||||
## Linear ticket
|
||||
|
||||
|
|
|
|||
|
|
@ -1610,6 +1610,7 @@ ALLOWED_VERTEX_AI_PASSTHROUGH_HEADERS: Final = {
|
|||
# e.g., 'x-pass-anthropic-beta: value' becomes 'anthropic-beta: value'
|
||||
# Works for all LLM pass-through endpoints (Vertex AI, Anthropic, Bedrock, etc.)
|
||||
PASS_THROUGH_HEADER_PREFIX: Final = "x-pass-"
|
||||
INTERNAL_KWARG_PREFIX: Final = "_litellm_"
|
||||
|
||||
AZURE_SPEECH_CUSTOM_LLM_PROVIDER: Final = "azure_speech"
|
||||
AZURE_SPEECH_PASS_THROUGH_ROUTE_PREFIX: Final = "/azure_speech"
|
||||
|
|
|
|||
|
|
@ -52,7 +52,7 @@ from litellm.types.router import GenericLiteLLMParams
|
|||
from litellm.types.utils import (
|
||||
LITELLM_IMAGE_VARIATION_PROVIDERS,
|
||||
LlmProviders,
|
||||
all_litellm_params,
|
||||
is_litellm_owned_kwarg,
|
||||
)
|
||||
from litellm.utils import (
|
||||
ImageResponse,
|
||||
|
|
@ -249,11 +249,9 @@ def image_generation(
|
|||
"size",
|
||||
"style",
|
||||
]
|
||||
litellm_params: Final = all_litellm_params
|
||||
default_params: Final = openai_params + litellm_params
|
||||
non_default_params: Final = {
|
||||
k: v for k, v in kwargs.items() if k not in default_params
|
||||
} # model-specific params - pass them straight to the model/provider
|
||||
k: v for k, v in kwargs.items() if k not in openai_params and not is_litellm_owned_kwarg(k)
|
||||
}
|
||||
|
||||
image_generation_config: BaseImageGenerationConfig | None = None
|
||||
if custom_llm_provider is not None and custom_llm_provider in LlmProviders._member_map_.values():
|
||||
|
|
@ -757,11 +755,9 @@ def image_edit(
|
|||
"style",
|
||||
"async_call",
|
||||
]
|
||||
litellm_params_list: Final = all_litellm_params
|
||||
default_params: Final = openai_params + litellm_params_list
|
||||
non_default_params: Final = {
|
||||
k: v for k, v in kwargs.items() if k not in default_params
|
||||
} # model-specific params - pass them straight to the model/provider
|
||||
k: v for k, v in kwargs.items() if k not in openai_params and not is_litellm_owned_kwarg(k)
|
||||
}
|
||||
litellm_logging_obj: Final[LiteLLMLoggingObj] = kwargs.get("litellm_logging_obj")
|
||||
litellm_call_id: Final[str | None] = kwargs.get("litellm_call_id", None)
|
||||
model_info: Final = kwargs.get("model_info", None)
|
||||
|
|
|
|||
|
|
@ -58,7 +58,7 @@ from litellm.types.llms.openai import (
|
|||
OpenAIFileObject,
|
||||
PathLike,
|
||||
)
|
||||
from litellm.types.utils import ExtractedFileData, LlmProviders, SpecialEnums, all_litellm_params
|
||||
from litellm.types.utils import ExtractedFileData, LlmProviders, SpecialEnums, is_litellm_owned_kwarg
|
||||
from litellm.utils import get_llm_provider, get_optional_params
|
||||
|
||||
from ..base_aws_llm import BaseAWSLLM
|
||||
|
|
@ -907,7 +907,7 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
{
|
||||
k: v
|
||||
for k, v in optional_params.items()
|
||||
if k not in all_litellm_params or k in _LITELLM_PARAMS_THE_MAPPER_TAKES
|
||||
if not is_litellm_owned_kwarg(k) or k in _LITELLM_PARAMS_THE_MAPPER_TAKES
|
||||
}
|
||||
),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -18,7 +18,7 @@ from litellm.llms.base_llm.text_to_speech.transformation import (
|
|||
TextToSpeechRequestData,
|
||||
)
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.utils import all_litellm_params
|
||||
from litellm.types.utils import is_litellm_owned_kwarg
|
||||
|
||||
from ..common_utils import ElevenLabsException
|
||||
|
||||
|
|
@ -241,7 +241,7 @@ class ElevenLabsTextToSpeechConfig(BaseTextToSpeechConfig):
|
|||
continue
|
||||
mapped_params[key] = value
|
||||
|
||||
reserved_kwarg_keys: Final = set(all_litellm_params) | {
|
||||
reserved_kwarg_keys: Final = {
|
||||
self.ELEVENLABS_QUERY_PARAMS_KEY,
|
||||
self.ELEVENLABS_VOICE_ID_KEY,
|
||||
"voice",
|
||||
|
|
@ -260,7 +260,7 @@ class ElevenLabsTextToSpeechConfig(BaseTextToSpeechConfig):
|
|||
mapped_params[key] = value
|
||||
|
||||
for key in list(kwargs.keys()):
|
||||
if key in reserved_kwarg_keys:
|
||||
if key in reserved_kwarg_keys or is_litellm_owned_kwarg(key):
|
||||
continue
|
||||
value = kwargs[key]
|
||||
if value is None:
|
||||
|
|
|
|||
|
|
@ -284,7 +284,7 @@ from .types.utils import (
|
|||
LlmProviders,
|
||||
PromptTokensDetails,
|
||||
ProviderSpecificHeader,
|
||||
all_litellm_params,
|
||||
is_litellm_owned_kwarg,
|
||||
)
|
||||
|
||||
####### ENVIRONMENT VARIABLES ###################
|
||||
|
|
@ -6351,15 +6351,10 @@ def embedding(
|
|||
"max_retries",
|
||||
"encoding_format",
|
||||
]
|
||||
litellm_params: Final = [
|
||||
"aembedding",
|
||||
"extra_headers",
|
||||
] + all_litellm_params
|
||||
|
||||
default_params: Final = openai_params + litellm_params
|
||||
default_params: Final = [*openai_params, "aembedding", "extra_headers"]
|
||||
non_default_params: Final = {
|
||||
k: v for k, v in kwargs.items() if k not in default_params
|
||||
} # model-specific params - pass them straight to the model/provider
|
||||
k: v for k, v in kwargs.items() if k not in default_params and not is_litellm_owned_kwarg(k)
|
||||
}
|
||||
|
||||
model, custom_llm_provider, dynamic_api_key, api_base = get_llm_provider(
|
||||
model=model,
|
||||
|
|
|
|||
|
|
@ -12216,7 +12216,8 @@
|
|||
"text",
|
||||
"image"
|
||||
],
|
||||
"supports_embedding_image_input": true
|
||||
"supports_embedding_image_input": true,
|
||||
"input_cost_per_image_token": 4.7e-07
|
||||
},
|
||||
"azure_ai/grok-4": {
|
||||
"input_cost_per_token": 3e-06,
|
||||
|
|
@ -15804,7 +15805,9 @@
|
|||
"mode": "embedding",
|
||||
"output_cost_per_token": 0.0,
|
||||
"output_vector_size": 1536,
|
||||
"supports_embedding_image_input": true
|
||||
"supports_embedding_image_input": true,
|
||||
"input_cost_per_image_token": 4.7e-07,
|
||||
"source": "https://cohere.com/pricing"
|
||||
},
|
||||
"cohere/parse-v5.0": {
|
||||
"litellm_provider": "cohere",
|
||||
|
|
@ -41913,14 +41916,14 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4-pro": {
|
||||
"cache_read_input_token_cost": 3.828e-08,
|
||||
"input_cost_per_token": 4.5936e-07,
|
||||
"cache_read_input_token_cost": 2.9e-08,
|
||||
"input_cost_per_token": 3.48e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 9.1872e-07,
|
||||
"output_cost_per_token": 6.96e-07,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -41933,14 +41936,14 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4.1-flash": {
|
||||
"cache_read_input_token_cost": 4.2e-09,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"cache_read_input_token_cost": 1e-09,
|
||||
"input_cost_per_token": 3.5e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.2e-07,
|
||||
"output_cost_per_token": 2.9e-07,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -43052,7 +43055,6 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/openai/gpt-oss-20b": {
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"input_cost_per_token": 1.8e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -43322,8 +43324,8 @@
|
|||
"input_cost_per_token": 2.6e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 65536,
|
||||
"max_tokens": 65536,
|
||||
"max_output_tokens": 235929,
|
||||
"max_tokens": 235929,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.08e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
|
|
@ -43603,8 +43605,8 @@
|
|||
"input_cost_per_token": 9.646e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 204800,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.0316e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
|
|
@ -66825,14 +66827,14 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/z-ai/glm-5.3-flash": {
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"input_cost_per_token": 4.5e-08,
|
||||
"cache_read_input_token_cost": 1.5e-08,
|
||||
"input_cost_per_token": 4e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1310720,
|
||||
"max_output_tokens": 943718,
|
||||
"max_tokens": 943718,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 5e-07,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -66865,13 +66867,13 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/z-ai/glm-5.3": {
|
||||
"input_cost_per_token": 1.4e-06,
|
||||
"output_cost_per_token": 4.4e-06,
|
||||
"cache_read_input_token_cost": 2.6e-07,
|
||||
"input_cost_per_token": 3.794e-07,
|
||||
"output_cost_per_token": 1.1924e-06,
|
||||
"cache_read_input_token_cost": 7.046e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1310720,
|
||||
"max_output_tokens": 943717,
|
||||
"max_tokens": 943717,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
|
|
@ -67520,8 +67522,8 @@
|
|||
"input_cost_per_token": 3.2e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262140,
|
||||
"max_tokens": 262140,
|
||||
"max_output_tokens": 81920,
|
||||
"max_tokens": 81920,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.2e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
|
|
@ -68297,8 +68299,8 @@
|
|||
"input_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
|
|
@ -68476,8 +68478,8 @@
|
|||
"cache_read_input_token_cost": 7e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"max_output_tokens": 235929,
|
||||
"max_tokens": 235929,
|
||||
"mode": "chat",
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
|
|
@ -68637,8 +68639,8 @@
|
|||
"output_cost_per_token": 3e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 32000,
|
||||
"max_tokens": 32000,
|
||||
"max_output_tokens": 235929,
|
||||
"max_tokens": 235929,
|
||||
"mode": "chat",
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
|
|
@ -69502,7 +69504,9 @@
|
|||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 128000,
|
||||
"max_tokens": 128000
|
||||
},
|
||||
"vertex_ai/gemini-2.5-flash-native-audio": {
|
||||
"deprecation_date": "2026-12-13",
|
||||
|
|
@ -69613,42 +69617,72 @@
|
|||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.5e-06,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 4096,
|
||||
"max_tokens": 4096
|
||||
},
|
||||
"together_ai/meta-llama/Llama-3.2-1B-Instruct": {
|
||||
"input_cost_per_token": 6e-08,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-08,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 131072,
|
||||
"max_tokens": 131072
|
||||
},
|
||||
"together_ai/meta-llama/Llama-3.2-3B-Instruct": {
|
||||
"input_cost_per_token": 6e-08,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-08,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 131072,
|
||||
"max_tokens": 131072
|
||||
},
|
||||
"together_ai/Qwen/Qwen2-1.5B-Instruct": {
|
||||
"input_cost_per_token": 2e-08,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2e-08,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 32768,
|
||||
"max_tokens": 32768
|
||||
},
|
||||
"together_ai/Qwen/Qwen2.5-14B-Instruct": {
|
||||
"input_cost_per_token": 8e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 8e-07,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 32768,
|
||||
"max_tokens": 32768
|
||||
},
|
||||
"together_ai/Qwen/Qwen2.5-72B-Instruct": {
|
||||
"input_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 32768,
|
||||
"max_tokens": 32768
|
||||
},
|
||||
"together_ai/Salesforce/Llama-Rank-V1": {
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"max_input_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "rerank",
|
||||
"output_cost_per_token": 0.0,
|
||||
"source": "https://api.together.xyz/v1/models"
|
||||
},
|
||||
"together_ai/meta-llama/Meta-Llama-3.1-8B": {
|
||||
"input_cost_per_token": 2e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"max_input_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"mode": "completion",
|
||||
"output_cost_per_token": 2e-07,
|
||||
"source": "https://api.together.xyz/v1/models"
|
||||
},
|
||||
"together_ai/together/Tev1-4B-experimental": {
|
||||
"cache_read_input_token_cost": 4.2e-08,
|
||||
|
|
|
|||
|
|
@ -2983,6 +2983,14 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
|
|||
"Set this well above health_check_interval because /health and the UI read the latest row per model."
|
||||
),
|
||||
)
|
||||
maximum_daily_tag_spend_retention_period: str | None = Field(
|
||||
None,
|
||||
description=(
|
||||
"Maximum retention period for per-day tag spend aggregate rows (e.g., '90d'). Rows whose day is older "
|
||||
"than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rows are never "
|
||||
"deleted. Only historical tag usage analytics are affected; tag budgets read the lifetime counter."
|
||||
),
|
||||
)
|
||||
use_spend_logs_partitioning: bool | None = Field(
|
||||
None,
|
||||
description="If True and LiteLLM_SpendLogs has been converted to a range-partitioned table (db_scripts/partition_spend_logs.sql), retention cleanup drops expired partitions instead of deleting rows, and pre-creates upcoming partitions. Default is False.",
|
||||
|
|
|
|||
|
|
@ -32,6 +32,17 @@ from litellm.proxy.utils import PrismaClient
|
|||
|
||||
StopReason: TypeAlias = Literal["exhausted", "budget_exhausted", "batch_cap_reached", "aborted"]
|
||||
|
||||
Cutoff: TypeAlias = datetime | str
|
||||
"""Rows strictly older than this are expired: a timestamp, or an ISO calendar day for tables keyed by day"""
|
||||
|
||||
|
||||
def _cutoff_cast(cutoff: Cutoff) -> str:
|
||||
return "timestamptz" if isinstance(cutoff, datetime) else "text"
|
||||
|
||||
|
||||
def _cutoff_text(cutoff: Cutoff) -> str:
|
||||
return cutoff.isoformat() if isinstance(cutoff, datetime) else cutoff
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class TableCleanupResult:
|
||||
|
|
@ -278,7 +289,7 @@ class SpendLogCleanup:
|
|||
return remaining
|
||||
|
||||
async def _execute_delete_batch(
|
||||
self, prisma_client: PrismaClient, delete_sql: str, cutoff_date: datetime, deadline: float
|
||||
self, prisma_client: PrismaClient, delete_sql: str, cutoff_date: Cutoff, deadline: float
|
||||
) -> int | None:
|
||||
"""
|
||||
Run one delete batch under a Postgres statement and lock timeout.
|
||||
|
|
@ -301,7 +312,7 @@ class SpendLogCleanup:
|
|||
return deleted_result if isinstance(deleted_result, int) else None
|
||||
|
||||
async def _count_remaining(
|
||||
self, prisma_client: PrismaClient, cutoff_date: datetime, table_name: str, time_column: str, deadline: float
|
||||
self, prisma_client: PrismaClient, cutoff_date: Cutoff, table_name: str, time_column: str, deadline: float
|
||||
) -> int | None:
|
||||
"""
|
||||
Count expired rows still outstanding, stopping at a cap.
|
||||
|
|
@ -314,7 +325,7 @@ class SpendLogCleanup:
|
|||
count_sql: Final = f"""
|
||||
SELECT count(*)::int AS remaining FROM (
|
||||
SELECT 1 FROM "{table_name}"
|
||||
WHERE "{time_column}" < $1::timestamptz
|
||||
WHERE "{time_column}" < $1::{_cutoff_cast(cutoff_date)}
|
||||
LIMIT $2
|
||||
) capped
|
||||
"""
|
||||
|
|
@ -332,7 +343,7 @@ class SpendLogCleanup:
|
|||
async def _delete_old_rows_batched(
|
||||
self,
|
||||
prisma_client: PrismaClient,
|
||||
cutoff_date: datetime,
|
||||
cutoff_date: Cutoff,
|
||||
table_name: str,
|
||||
key_columns: tuple[str, ...],
|
||||
time_column: str,
|
||||
|
|
@ -350,7 +361,7 @@ class SpendLogCleanup:
|
|||
DELETE FROM "{table_name}"
|
||||
WHERE ({key_list}) IN (
|
||||
SELECT {key_list} FROM "{table_name}"
|
||||
WHERE "{time_column}" < $1::timestamptz
|
||||
WHERE "{time_column}" < $1::{_cutoff_cast(cutoff_date)}
|
||||
LIMIT $2
|
||||
)
|
||||
"""
|
||||
|
|
@ -406,7 +417,7 @@ class SpendLogCleanup:
|
|||
run_count,
|
||||
consecutive_failures,
|
||||
self.batch_size,
|
||||
cutoff_date.isoformat(),
|
||||
_cutoff_text(cutoff_date),
|
||||
total_deleted,
|
||||
type(batch_exc).__name__,
|
||||
batch_exc,
|
||||
|
|
@ -454,7 +465,7 @@ class SpendLogCleanup:
|
|||
async def _finish_table(
|
||||
self,
|
||||
prisma_client: PrismaClient,
|
||||
cutoff_date: datetime,
|
||||
cutoff_date: Cutoff,
|
||||
table_name: str,
|
||||
time_column: str,
|
||||
rows_deleted: int,
|
||||
|
|
@ -541,6 +552,18 @@ class SpendLogCleanup:
|
|||
deadline=deadline,
|
||||
)
|
||||
|
||||
async def _delete_old_daily_tag_spend_rows(
|
||||
self, prisma_client: PrismaClient, cutoff_day: str, deadline: float
|
||||
) -> TableCleanupResult:
|
||||
return await self._delete_old_rows_batched(
|
||||
prisma_client,
|
||||
cutoff_day,
|
||||
table_name="LiteLLM_DailyTagSpend",
|
||||
key_columns=("id",),
|
||||
time_column="date",
|
||||
deadline=deadline,
|
||||
)
|
||||
|
||||
async def _clean_spend_log_tables(
|
||||
self, prisma_client: PrismaClient, deadline: float
|
||||
) -> tuple[TableCleanupResult, ...]:
|
||||
|
|
@ -624,6 +647,18 @@ class SpendLogCleanup:
|
|||
)
|
||||
return (health_checks_result,)
|
||||
|
||||
async def _clean_daily_tag_spend(
|
||||
self, prisma_client: PrismaClient, retention_seconds: int, deadline: float
|
||||
) -> tuple[TableCleanupResult, ...]:
|
||||
"""
|
||||
Prune per-day tag spend rows whose ISO day sorts before the horizon day; the horizon day itself is kept.
|
||||
"""
|
||||
horizon: Final = datetime.now(timezone.utc) - timedelta(seconds=float(retention_seconds))
|
||||
cutoff_day: Final = horizon.date().isoformat()
|
||||
result: Final = await self._delete_old_daily_tag_spend_rows(prisma_client, cutoff_day, deadline)
|
||||
verbose_proxy_logger.info("Deleted %s expired daily tag spend rows", result.rows_deleted)
|
||||
return (result,)
|
||||
|
||||
@staticmethod
|
||||
def _run_outcome(results: tuple[TableCleanupResult, ...]) -> RunOutcome:
|
||||
"""
|
||||
|
|
@ -671,10 +706,14 @@ class SpendLogCleanup:
|
|||
"maximum_autorouter_session_retention_period"
|
||||
)
|
||||
health_check_retention_seconds: Final = self._retention_seconds_for("maximum_health_check_retention_period")
|
||||
daily_tag_spend_retention_seconds: Final = self._retention_seconds_for(
|
||||
"maximum_daily_tag_spend_retention_period"
|
||||
)
|
||||
if (
|
||||
not delete_spend_logs
|
||||
and autorouter_retention_seconds is None
|
||||
and health_check_retention_seconds is None
|
||||
and daily_tag_spend_retention_seconds is None
|
||||
):
|
||||
SpendLogCleanupMetrics.record_run("skipped_disabled")
|
||||
return
|
||||
|
|
@ -706,6 +745,7 @@ class SpendLogCleanup:
|
|||
int(delete_spend_logs and self.retention_seconds is not None)
|
||||
+ int(autorouter_retention_seconds is not None)
|
||||
+ int(health_check_retention_seconds is not None)
|
||||
+ int(daily_tag_spend_retention_seconds is not None)
|
||||
)
|
||||
|
||||
spend_log_results: Final = (
|
||||
|
|
@ -716,8 +756,13 @@ class SpendLogCleanup:
|
|||
if delete_spend_logs and self.retention_seconds is not None
|
||||
else ()
|
||||
)
|
||||
remaining_groups_after_spend_logs: Final = int(autorouter_retention_seconds is not None) + int(
|
||||
health_check_retention_seconds is not None
|
||||
remaining_groups_after_spend_logs: Final = (
|
||||
int(autorouter_retention_seconds is not None)
|
||||
+ int(health_check_retention_seconds is not None)
|
||||
+ int(daily_tag_spend_retention_seconds is not None)
|
||||
)
|
||||
remaining_groups_after_sessions: Final = int(health_check_retention_seconds is not None) + int(
|
||||
daily_tag_spend_retention_seconds is not None
|
||||
)
|
||||
session_results: Final = (
|
||||
await self._clean_session_rollup(
|
||||
|
|
@ -732,13 +777,18 @@ class SpendLogCleanup:
|
|||
await self._clean_health_checks(
|
||||
prisma_client,
|
||||
health_check_retention_seconds,
|
||||
deadline,
|
||||
self._group_deadline(deadline, remaining_groups_after_sessions),
|
||||
)
|
||||
if health_check_retention_seconds is not None
|
||||
else ()
|
||||
)
|
||||
daily_tag_spend_results: Final = (
|
||||
await self._clean_daily_tag_spend(prisma_client, daily_tag_spend_retention_seconds, deadline)
|
||||
if daily_tag_spend_retention_seconds is not None
|
||||
else ()
|
||||
)
|
||||
|
||||
results: Final = spend_log_results + session_results + health_check_results
|
||||
results: Final = spend_log_results + session_results + health_check_results + daily_tag_spend_results
|
||||
outcome: Final = self._run_outcome(results)
|
||||
SpendLogCleanupMetrics.record_run(outcome)
|
||||
self._log_run_summary(outcome, results, time.monotonic() - run_started_at)
|
||||
|
|
|
|||
|
|
@ -213,6 +213,8 @@ try:
|
|||
import orjson
|
||||
import yaml
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
from apscheduler.schedulers.base import STATE_STOPPED
|
||||
from apscheduler.triggers.base import BaseTrigger
|
||||
from apscheduler.triggers.interval import IntervalTrigger
|
||||
except ImportError as e:
|
||||
raise ImportError(f"Missing dependency {e}. Run `pip install 'litellm[proxy]'`")
|
||||
|
|
@ -5078,6 +5080,20 @@ def _current_general_settings() -> Mapping[str, object]:
|
|||
return general_settings
|
||||
|
||||
|
||||
_CLEANUP_SCHEDULE_KEYS: Final = (
|
||||
"maximum_spend_logs_retention_period",
|
||||
"maximum_autorouter_session_retention_period",
|
||||
"maximum_health_check_retention_period",
|
||||
"maximum_daily_tag_spend_retention_period",
|
||||
"maximum_spend_logs_cleanup_cron",
|
||||
"maximum_spend_logs_retention_interval",
|
||||
)
|
||||
|
||||
|
||||
def _cleanup_schedule_of(settings: Mapping[str, object]) -> tuple[object, ...]:
|
||||
return tuple(settings.get(key) for key in _CLEANUP_SCHEDULE_KEYS)
|
||||
|
||||
|
||||
@lru_cache(maxsize=4096)
|
||||
def _log_ignored_cost_map_copy(model_id: str, fields: tuple[str, ...]) -> None:
|
||||
verbose_proxy_logger.warning(
|
||||
|
|
@ -5100,6 +5116,8 @@ class ProxyConfig:
|
|||
self._last_websearch_interception_config: dict[str, object] | None = None
|
||||
self._last_hashicorp_vault_config: dict[str, object] | None = None
|
||||
self._last_cyberark_config: dict[str, object] | None = None # mutable-ok: change-detection cache
|
||||
self._last_cleanup_schedule_attempt: tuple[object, ...] | None = None
|
||||
self._cleanup_reschedule_failed: bool = False
|
||||
self._cyberark_boot_env: dict[str, str | None] | None = None # mutable-ok: deployment env snapshot, set once
|
||||
self.worker_registry: list[WorkerRegistryEntry] = []
|
||||
self.config_sync_subscriber: ConfigSyncSubscriber | None = None
|
||||
|
|
@ -7455,69 +7473,67 @@ class ProxyConfig:
|
|||
if scheduler is None:
|
||||
return
|
||||
|
||||
# Remove existing job if it exists
|
||||
try:
|
||||
scheduler.remove_job("spend_log_cleanup_job")
|
||||
verbose_proxy_logger.info("Removed existing spend log cleanup job")
|
||||
except Exception:
|
||||
pass # Job might not exist, which is fine
|
||||
|
||||
# Schedule new job if retention period is set (not None)
|
||||
retention_period: Final = general_settings.get("maximum_spend_logs_retention_period")
|
||||
autorouter_retention: Final = general_settings.get("maximum_autorouter_session_retention_period")
|
||||
health_check_retention: Final = general_settings.get("maximum_health_check_retention_period")
|
||||
if retention_period is not None or autorouter_retention is not None or health_check_retention is not None:
|
||||
from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import (
|
||||
SpendLogCleanup,
|
||||
wants_job: Final = any(
|
||||
general_settings.get(key) is not None
|
||||
for key in (
|
||||
"maximum_spend_logs_retention_period",
|
||||
"maximum_autorouter_session_retention_period",
|
||||
"maximum_health_check_retention_period",
|
||||
"maximum_daily_tag_spend_retention_period",
|
||||
)
|
||||
)
|
||||
if not wants_job:
|
||||
if scheduler.get_job("spend_log_cleanup_job") is not None:
|
||||
scheduler.remove_job("spend_log_cleanup_job")
|
||||
verbose_proxy_logger.info("Removed existing spend log cleanup job")
|
||||
return
|
||||
|
||||
spend_log_cleanup: Final = SpendLogCleanup()
|
||||
cleanup_cron: Final = general_settings.get("maximum_spend_logs_cleanup_cron")
|
||||
trigger: Final = self._spend_log_cleanup_trigger()
|
||||
if trigger is None:
|
||||
return
|
||||
from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import (
|
||||
SpendLogCleanup,
|
||||
)
|
||||
|
||||
if cleanup_cron:
|
||||
from apscheduler.triggers.cron import CronTrigger
|
||||
scheduler.add_job(
|
||||
SpendLogCleanup().cleanup_old_spend_logs,
|
||||
trigger,
|
||||
args=[prisma_client],
|
||||
id="spend_log_cleanup_job",
|
||||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
verbose_proxy_logger.info("Spend log cleanup rescheduled with trigger: %s", trigger)
|
||||
|
||||
try:
|
||||
cron_trigger: Final = CronTrigger.from_crontab(cleanup_cron)
|
||||
scheduler.add_job(
|
||||
spend_log_cleanup.cleanup_old_spend_logs,
|
||||
cron_trigger,
|
||||
args=[prisma_client],
|
||||
id="spend_log_cleanup_job",
|
||||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
verbose_proxy_logger.info("Spend log cleanup rescheduled with cron: %s", cleanup_cron)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.error("Invalid maximum_spend_logs_cleanup_cron value: %s", cleanup_cron)
|
||||
else:
|
||||
# Interval-based scheduling (existing behavior)
|
||||
from litellm.litellm_core_utils.duration_parser import (
|
||||
duration_in_seconds,
|
||||
)
|
||||
def _spend_log_cleanup_trigger(self) -> BaseTrigger | None:
|
||||
cleanup_cron: Final[object] = general_settings.get("maximum_spend_logs_cleanup_cron")
|
||||
if cleanup_cron:
|
||||
from apscheduler.triggers.cron import CronTrigger
|
||||
|
||||
retention_interval: Final = general_settings.get("maximum_spend_logs_retention_interval", "1d")
|
||||
try:
|
||||
interval_seconds: Final = duration_in_seconds(retention_interval)
|
||||
# this runs against a started scheduler, which the startup stagger sweep
|
||||
# cannot reach, so the offset is applied here or the job reconverges across
|
||||
# replicas the first time an admin edits the retention settings
|
||||
scheduler.add_job(
|
||||
spend_log_cleanup.cleanup_old_spend_logs,
|
||||
stagger_trigger(
|
||||
job_id="spend_log_cleanup_job",
|
||||
trigger=IntervalTrigger(seconds=interval_seconds),
|
||||
period_seconds=interval_seconds,
|
||||
settings=parse_stagger_settings(general_settings),
|
||||
),
|
||||
args=[prisma_client],
|
||||
id="spend_log_cleanup_job",
|
||||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
verbose_proxy_logger.info("Spend log cleanup rescheduled with interval: %s", retention_interval)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.error("Invalid maximum_spend_logs_retention_interval value")
|
||||
try:
|
||||
cron_trigger: Final[BaseTrigger] = CronTrigger.from_crontab(cleanup_cron)
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
verbose_proxy_logger.error("Invalid maximum_spend_logs_cleanup_cron value: %s", cleanup_cron)
|
||||
return None
|
||||
return cron_trigger
|
||||
retention_interval: Final[object] = general_settings.get("maximum_spend_logs_retention_interval", "1d")
|
||||
if not isinstance(retention_interval, str):
|
||||
verbose_proxy_logger.error("Invalid maximum_spend_logs_retention_interval value: %r", retention_interval)
|
||||
return None
|
||||
# this runs against a started scheduler, which the startup stagger sweep
|
||||
# cannot reach, so the offset is applied here or the job reconverges across
|
||||
# replicas the first time an admin edits the retention settings
|
||||
try:
|
||||
interval_seconds: Final = duration_in_seconds(retention_interval)
|
||||
return stagger_trigger(
|
||||
job_id="spend_log_cleanup_job",
|
||||
trigger=IntervalTrigger(seconds=interval_seconds),
|
||||
period_seconds=interval_seconds,
|
||||
settings=parse_stagger_settings(general_settings),
|
||||
)
|
||||
except (ValueError, OverflowError):
|
||||
verbose_proxy_logger.error("Invalid maximum_spend_logs_retention_interval value: %r", retention_interval)
|
||||
return None
|
||||
|
||||
async def _update_general_settings(self, db_general_settings: Mapping[str, SettingsJsonValue] | None) -> None:
|
||||
global general_settings
|
||||
|
|
@ -7526,32 +7542,28 @@ class ProxyConfig:
|
|||
if not isinstance(general_settings, SettingsStore):
|
||||
self.settings.load_yaml(_as_settings_mapping(general_settings))
|
||||
cache_size_was_db: Final = self.settings.source("user_api_key_cache_max_size") == "db"
|
||||
previous_retention_values: Final = self._resolved_retention_values()
|
||||
previous_cleanup_schedule: Final = self._resolved_cleanup_schedule()
|
||||
previous_pass_through_endpoints: Final = self.settings.get("pass_through_endpoints")
|
||||
self.settings.apply_db_row("general_settings", db_general_settings)
|
||||
_bind_general_settings_store(self.settings)
|
||||
await self._apply_general_settings_side_effects(
|
||||
db_general_settings,
|
||||
cache_size_was_db,
|
||||
previous_retention_values,
|
||||
previous_cleanup_schedule,
|
||||
previous_pass_through_endpoints,
|
||||
)
|
||||
|
||||
def _resolved_retention_values(self) -> tuple[SettingsJsonValue | None, ...]:
|
||||
return tuple(
|
||||
self.settings.get(key)
|
||||
for key in (
|
||||
"maximum_spend_logs_retention_period",
|
||||
"maximum_autorouter_session_retention_period",
|
||||
"maximum_health_check_retention_period",
|
||||
)
|
||||
)
|
||||
def _resolved_cleanup_schedule(self) -> tuple[object, ...]:
|
||||
return _cleanup_schedule_of(self.settings)
|
||||
|
||||
def record_cleanup_schedule_attempt(self, settings: Mapping[str, object]) -> None:
|
||||
self._last_cleanup_schedule_attempt = _cleanup_schedule_of(settings)
|
||||
|
||||
async def _apply_general_settings_side_effects(
|
||||
self,
|
||||
db_values: Mapping[str, SettingsJsonValue],
|
||||
cache_size_was_db: bool,
|
||||
previous_retention_values: tuple[SettingsJsonValue | None, ...],
|
||||
previous_cleanup_schedule: tuple[object, ...],
|
||||
previous_pass_through_endpoints: SettingsJsonValue | None,
|
||||
) -> None:
|
||||
effects: Final = (
|
||||
|
|
@ -7560,7 +7572,7 @@ class ProxyConfig:
|
|||
self._apply_boolean_settings,
|
||||
partial(self._apply_cache_size_setting, cache_size_was_db=cache_size_was_db),
|
||||
self._apply_store_model_in_db_setting,
|
||||
partial(self._apply_retention_settings, previous_retention_values=previous_retention_values),
|
||||
partial(self._apply_retention_settings, previous_cleanup_schedule=previous_cleanup_schedule),
|
||||
self._apply_ssrf_settings,
|
||||
)
|
||||
for effect in effects:
|
||||
|
|
@ -7655,10 +7667,36 @@ class ProxyConfig:
|
|||
async def _apply_retention_settings(
|
||||
self,
|
||||
db_values: Mapping[str, SettingsJsonValue],
|
||||
previous_retention_values: tuple[SettingsJsonValue | None, ...],
|
||||
previous_cleanup_schedule: tuple[object, ...],
|
||||
) -> None:
|
||||
if previous_retention_values != self._resolved_retention_values():
|
||||
# while the scheduler is still stopped the startup block owns the first registration
|
||||
if scheduler is not None and scheduler.state == STATE_STOPPED:
|
||||
return
|
||||
schedule: Final = self._resolved_cleanup_schedule()
|
||||
wants_job: Final = any(value is not None for value in schedule[:4])
|
||||
has_job: Final = scheduler is not None and scheduler.get_job("spend_log_cleanup_job") is not None
|
||||
baseline: Final = (
|
||||
self._last_cleanup_schedule_attempt
|
||||
if has_job and self._last_cleanup_schedule_attempt is not None
|
||||
else previous_cleanup_schedule
|
||||
)
|
||||
retry_due: Final = (
|
||||
wants_job
|
||||
and (not has_job or self._cleanup_reschedule_failed)
|
||||
and schedule != self._last_cleanup_schedule_attempt
|
||||
)
|
||||
if not (baseline != schedule or retry_due or (has_job and not wants_job)):
|
||||
return
|
||||
try:
|
||||
await self._reschedule_spend_log_cleanup_job()
|
||||
except Exception as exc:
|
||||
self._cleanup_reschedule_failed = True
|
||||
verbose_proxy_logger.exception(
|
||||
"Spend log cleanup could not be rescheduled, will retry on next sync: %s", exc
|
||||
)
|
||||
return
|
||||
self._cleanup_reschedule_failed = False
|
||||
self._last_cleanup_schedule_attempt = schedule
|
||||
|
||||
async def _apply_ssrf_settings(self, db_values: Mapping[str, SettingsJsonValue]) -> None:
|
||||
_apply_ssrf_general_settings(db_values)
|
||||
|
|
@ -10535,15 +10573,19 @@ class ProxyStartupEvent:
|
|||
)
|
||||
|
||||
### SPEND LOG CLEANUP ###
|
||||
cleanup_settings: Final = _current_general_settings()
|
||||
if (
|
||||
general_settings.get("maximum_spend_logs_retention_period") is not None
|
||||
or general_settings.get("maximum_autorouter_session_retention_period") is not None
|
||||
or general_settings.get("maximum_health_check_retention_period") is not None
|
||||
cleanup_settings.get("maximum_spend_logs_retention_period") is not None
|
||||
or cleanup_settings.get("maximum_autorouter_session_retention_period") is not None
|
||||
or cleanup_settings.get("maximum_health_check_retention_period") is not None
|
||||
or cleanup_settings.get("maximum_daily_tag_spend_retention_period") is not None
|
||||
):
|
||||
spend_log_cleanup: Final = SpendLogCleanup()
|
||||
cleanup_cron: Final = general_settings.get("maximum_spend_logs_cleanup_cron")
|
||||
cleanup_cron: Final = cleanup_settings.get("maximum_spend_logs_cleanup_cron")
|
||||
|
||||
if cleanup_cron:
|
||||
if cleanup_cron and not isinstance(cleanup_cron, str):
|
||||
verbose_proxy_logger.error("Invalid maximum_spend_logs_cleanup_cron value: %r", cleanup_cron)
|
||||
elif isinstance(cleanup_cron, str) and cleanup_cron:
|
||||
from apscheduler.triggers.cron import CronTrigger
|
||||
|
||||
try:
|
||||
|
|
@ -10561,8 +10603,10 @@ class ProxyStartupEvent:
|
|||
verbose_proxy_logger.error("Invalid maximum_spend_logs_cleanup_cron value: %s", cleanup_cron)
|
||||
else:
|
||||
# Interval-based scheduling (existing behavior)
|
||||
retention_interval: Final = general_settings.get("maximum_spend_logs_retention_interval", "1d")
|
||||
retention_interval: Final = cleanup_settings.get("maximum_spend_logs_retention_interval", "1d")
|
||||
try:
|
||||
if not isinstance(retention_interval, str):
|
||||
raise ValueError(retention_interval)
|
||||
interval_seconds: Final = duration_in_seconds(retention_interval)
|
||||
scheduler.add_job(
|
||||
spend_log_cleanup.cleanup_old_spend_logs,
|
||||
|
|
@ -10573,8 +10617,11 @@ class ProxyStartupEvent:
|
|||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.error("Invalid maximum_spend_logs_retention_interval value")
|
||||
except (ValueError, OverflowError):
|
||||
verbose_proxy_logger.error(
|
||||
"Invalid maximum_spend_logs_retention_interval value: %r", retention_interval
|
||||
)
|
||||
proxy_config.record_cleanup_schedule_attempt(cleanup_settings)
|
||||
### CHECK BATCH COST ###
|
||||
if llm_router is not None and PROXY_BATCH_POLLING_ENABLED:
|
||||
try:
|
||||
|
|
@ -17909,6 +17956,7 @@ _GENERAL_SETTINGS_CONFIG_LIST_FIELD_TYPES: Final[Mapping[str, str]] = MappingPro
|
|||
"store_prompts_in_spend_logs": "Boolean",
|
||||
"maximum_spend_logs_retention_period": "String",
|
||||
"maximum_health_check_retention_period": "String",
|
||||
"maximum_daily_tag_spend_retention_period": "String",
|
||||
"maximum_spend_logs_cleanup_batch_size": "Integer",
|
||||
"maximum_spend_logs_cleanup_max_batches": "Integer",
|
||||
"maximum_spend_logs_cleanup_run_budget": "String",
|
||||
|
|
|
|||
|
|
@ -48,6 +48,7 @@ from typing_extensions import NotRequired, ReadOnly, Required, TypedDict
|
|||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm._uuid import uuid
|
||||
from litellm.constants import INTERNAL_KWARG_PREFIX
|
||||
from litellm.types.llms.base import (
|
||||
BaseLiteLLMOpenAIResponseObject,
|
||||
CachedTokensDetails,
|
||||
|
|
@ -3937,6 +3938,10 @@ all_litellm_params = [ # rebind-ok: two star imports in litellm/__init__.py re-
|
|||
]
|
||||
|
||||
|
||||
def is_litellm_owned_kwarg(name: str) -> bool:
|
||||
return name in all_litellm_params or name.startswith(INTERNAL_KWARG_PREFIX)
|
||||
|
||||
|
||||
class KeyGenerationConfig(TypedDict, total=False):
|
||||
required_params: list[str] # specify params that must be present in the key generation request
|
||||
|
||||
|
|
|
|||
|
|
@ -257,7 +257,7 @@ from litellm.types.utils import (
|
|||
TextCompletionResponse,
|
||||
TranscriptionResponse,
|
||||
Usage,
|
||||
all_litellm_params,
|
||||
is_litellm_owned_kwarg,
|
||||
)
|
||||
|
||||
_CALL_TYPE_ENUM_MAP: Final[dict] = {ct.value: ct for ct in CallTypes}
|
||||
|
|
@ -4161,26 +4161,8 @@ def _remove_unsupported_params(non_default_params: dict, supported_openai_params
|
|||
return non_default_params
|
||||
|
||||
|
||||
def filter_out_litellm_params(kwargs: dict) -> dict:
|
||||
"""
|
||||
Filter out LiteLLM internal parameters from kwargs dict.
|
||||
|
||||
Returns a new dict containing only non-LiteLLM parameters that should be
|
||||
passed to external provider APIs.
|
||||
|
||||
Args:
|
||||
kwargs: Dictionary that may contain LiteLLM internal parameters
|
||||
|
||||
Returns:
|
||||
Dictionary with LiteLLM internal parameters filtered out
|
||||
|
||||
Example:
|
||||
>>> kwargs = {"query": "test", "shared_session": session_obj, "metadata": {}}
|
||||
>>> filtered = filter_out_litellm_params(kwargs)
|
||||
>>> # filtered = {"query": "test"}
|
||||
"""
|
||||
|
||||
return {key: value for key, value in kwargs.items() if key not in all_litellm_params}
|
||||
def filter_out_litellm_params(kwargs: Mapping[str, object]) -> dict:
|
||||
return {key: value for key, value in kwargs.items() if not is_litellm_owned_kwarg(key)}
|
||||
|
||||
|
||||
def _provider_supports_vertex_params(custom_llm_provider: str) -> bool:
|
||||
|
|
@ -10152,10 +10134,9 @@ def get_standard_openai_params(params: Mapping[str, object]) -> dict:
|
|||
|
||||
def get_non_default_completion_params(kwargs: Mapping[str, object]) -> dict:
|
||||
openai_params: Final = litellm.OPENAI_CHAT_COMPLETION_PARAMS
|
||||
default_params: Final = openai_params + all_litellm_params
|
||||
non_default_params: Final = {
|
||||
k: v for k, v in kwargs.items() if k not in default_params
|
||||
} # model-specific params - pass them straight to the model/provider
|
||||
k: v for k, v in kwargs.items() if k not in openai_params and not is_litellm_owned_kwarg(k)
|
||||
}
|
||||
|
||||
return non_default_params
|
||||
|
||||
|
|
@ -10203,11 +10184,12 @@ def strip_reasoning_summary_aliases_from_optional_params(
|
|||
return op, rs_val
|
||||
|
||||
|
||||
def get_non_default_transcription_params(kwargs: dict) -> dict:
|
||||
def get_non_default_transcription_params(kwargs: Mapping[str, object]) -> dict:
|
||||
from litellm.constants import OPENAI_TRANSCRIPTION_PARAMS
|
||||
|
||||
default_params: Final = OPENAI_TRANSCRIPTION_PARAMS + all_litellm_params
|
||||
non_default_params: Final = {k: v for k, v in kwargs.items() if k not in default_params}
|
||||
non_default_params: Final = {
|
||||
k: v for k, v in kwargs.items() if k not in OPENAI_TRANSCRIPTION_PARAMS and not is_litellm_owned_kwarg(k)
|
||||
}
|
||||
return non_default_params
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -12216,7 +12216,8 @@
|
|||
"text",
|
||||
"image"
|
||||
],
|
||||
"supports_embedding_image_input": true
|
||||
"supports_embedding_image_input": true,
|
||||
"input_cost_per_image_token": 4.7e-07
|
||||
},
|
||||
"azure_ai/grok-4": {
|
||||
"input_cost_per_token": 3e-06,
|
||||
|
|
@ -15804,7 +15805,9 @@
|
|||
"mode": "embedding",
|
||||
"output_cost_per_token": 0.0,
|
||||
"output_vector_size": 1536,
|
||||
"supports_embedding_image_input": true
|
||||
"supports_embedding_image_input": true,
|
||||
"input_cost_per_image_token": 4.7e-07,
|
||||
"source": "https://cohere.com/pricing"
|
||||
},
|
||||
"cohere/parse-v5.0": {
|
||||
"litellm_provider": "cohere",
|
||||
|
|
@ -41913,14 +41916,14 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4-pro": {
|
||||
"cache_read_input_token_cost": 3.828e-08,
|
||||
"input_cost_per_token": 4.5936e-07,
|
||||
"cache_read_input_token_cost": 2.9e-08,
|
||||
"input_cost_per_token": 3.48e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 9.1872e-07,
|
||||
"output_cost_per_token": 6.96e-07,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -41933,14 +41936,14 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4.1-flash": {
|
||||
"cache_read_input_token_cost": 4.2e-09,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"cache_read_input_token_cost": 1e-09,
|
||||
"input_cost_per_token": 3.5e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.2e-07,
|
||||
"output_cost_per_token": 2.9e-07,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -43052,7 +43055,6 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/openai/gpt-oss-20b": {
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"input_cost_per_token": 1.8e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -43322,8 +43324,8 @@
|
|||
"input_cost_per_token": 2.6e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 65536,
|
||||
"max_tokens": 65536,
|
||||
"max_output_tokens": 235929,
|
||||
"max_tokens": 235929,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.08e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
|
|
@ -43603,8 +43605,8 @@
|
|||
"input_cost_per_token": 9.646e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 204800,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.0316e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
|
|
@ -66825,14 +66827,14 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/z-ai/glm-5.3-flash": {
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"input_cost_per_token": 4.5e-08,
|
||||
"cache_read_input_token_cost": 1.5e-08,
|
||||
"input_cost_per_token": 4e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1310720,
|
||||
"max_output_tokens": 943718,
|
||||
"max_tokens": 943718,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 5e-07,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -66865,13 +66867,13 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/z-ai/glm-5.3": {
|
||||
"input_cost_per_token": 1.4e-06,
|
||||
"output_cost_per_token": 4.4e-06,
|
||||
"cache_read_input_token_cost": 2.6e-07,
|
||||
"input_cost_per_token": 3.794e-07,
|
||||
"output_cost_per_token": 1.1924e-06,
|
||||
"cache_read_input_token_cost": 7.046e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1310720,
|
||||
"max_output_tokens": 943717,
|
||||
"max_tokens": 943717,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
|
|
@ -67520,8 +67522,8 @@
|
|||
"input_cost_per_token": 3.2e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262140,
|
||||
"max_tokens": 262140,
|
||||
"max_output_tokens": 81920,
|
||||
"max_tokens": 81920,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.2e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
|
|
@ -68297,8 +68299,8 @@
|
|||
"input_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
|
|
@ -68476,8 +68478,8 @@
|
|||
"cache_read_input_token_cost": 7e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"max_output_tokens": 235929,
|
||||
"max_tokens": 235929,
|
||||
"mode": "chat",
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
|
|
@ -68637,8 +68639,8 @@
|
|||
"output_cost_per_token": 3e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 32000,
|
||||
"max_tokens": 32000,
|
||||
"max_output_tokens": 235929,
|
||||
"max_tokens": 235929,
|
||||
"mode": "chat",
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
|
|
@ -69502,7 +69504,9 @@
|
|||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 128000,
|
||||
"max_tokens": 128000
|
||||
},
|
||||
"vertex_ai/gemini-2.5-flash-native-audio": {
|
||||
"deprecation_date": "2026-12-13",
|
||||
|
|
@ -69613,42 +69617,72 @@
|
|||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.5e-06,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 4096,
|
||||
"max_tokens": 4096
|
||||
},
|
||||
"together_ai/meta-llama/Llama-3.2-1B-Instruct": {
|
||||
"input_cost_per_token": 6e-08,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-08,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 131072,
|
||||
"max_tokens": 131072
|
||||
},
|
||||
"together_ai/meta-llama/Llama-3.2-3B-Instruct": {
|
||||
"input_cost_per_token": 6e-08,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-08,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 131072,
|
||||
"max_tokens": 131072
|
||||
},
|
||||
"together_ai/Qwen/Qwen2-1.5B-Instruct": {
|
||||
"input_cost_per_token": 2e-08,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2e-08,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 32768,
|
||||
"max_tokens": 32768
|
||||
},
|
||||
"together_ai/Qwen/Qwen2.5-14B-Instruct": {
|
||||
"input_cost_per_token": 8e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 8e-07,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 32768,
|
||||
"max_tokens": 32768
|
||||
},
|
||||
"together_ai/Qwen/Qwen2.5-72B-Instruct": {
|
||||
"input_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"source": "https://api.together.ai/v1/models"
|
||||
"source": "https://api.together.ai/v1/models",
|
||||
"max_input_tokens": 32768,
|
||||
"max_tokens": 32768
|
||||
},
|
||||
"together_ai/Salesforce/Llama-Rank-V1": {
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"max_input_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "rerank",
|
||||
"output_cost_per_token": 0.0,
|
||||
"source": "https://api.together.xyz/v1/models"
|
||||
},
|
||||
"together_ai/meta-llama/Meta-Llama-3.1-8B": {
|
||||
"input_cost_per_token": 2e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"max_input_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"mode": "completion",
|
||||
"output_cost_per_token": 2e-07,
|
||||
"source": "https://api.together.xyz/v1/models"
|
||||
},
|
||||
"together_ai/together/Tev1-4B-experimental": {
|
||||
"cache_read_input_token_cost": 4.2e-08,
|
||||
|
|
|
|||
|
|
@ -276,7 +276,7 @@ def provider_wire_environment(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -
|
|||
@pytest.mark.parametrize("provider", PROVIDERS)
|
||||
@pytest.mark.parametrize("asynchronous", [False, True])
|
||||
@pytest.mark.parametrize("stream", [False, True])
|
||||
async def test_stream_chunk_size_never_reaches_provider_body(
|
||||
async def test_internal_params_never_reach_provider_body(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
provider_wire_environment: None,
|
||||
provider: str,
|
||||
|
|
@ -289,6 +289,7 @@ async def test_stream_chunk_size_never_reaches_provider_body(
|
|||
**_request_parameters(provider, wire.url),
|
||||
"stream": stream,
|
||||
"stream_chunk_size": 64,
|
||||
"_litellm_undeclared_sentinel": "internal",
|
||||
"extra_body": {"custom_provider_key": 1},
|
||||
"max_tokens": 16,
|
||||
"timeout": 5,
|
||||
|
|
@ -313,4 +314,5 @@ async def test_stream_chunk_size_never_reaches_provider_body(
|
|||
keys: Final = keys_at_every_depth(body)
|
||||
assert "stream_chunk_size" not in keys
|
||||
assert not INTERNAL_FIELDS.intersection(keys)
|
||||
assert not frozenset(key for key in keys if key.startswith("_litellm_")), keys
|
||||
assert _custom_key(body, provider) == 1
|
||||
247
tests/integration/spend/test_daily_tag_spend_retention.py
Normal file
247
tests/integration/spend/test_daily_tag_spend_retention.py
Normal file
|
|
@ -0,0 +1,247 @@
|
|||
import json
|
||||
import os
|
||||
import signal
|
||||
import uuid
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import psutil
|
||||
import psycopg
|
||||
import pytest
|
||||
import yaml
|
||||
from pydantic import JsonValue, TypeAdapter
|
||||
|
||||
from tests.integration._support.client import Gateway, eventually, string_value
|
||||
from tests.integration._support.database import read_rows
|
||||
from tests.integration._support.process import OwnedProxy, owned_proxy, owned_proxy_process
|
||||
|
||||
CLEANUP_EVERY_MINUTE: Final = "* * * * *"
|
||||
RETENTION_SETTING: Final = "maximum_daily_tag_spend_retention_period"
|
||||
_MAPPING: Final = TypeAdapter(dict[str, JsonValue])
|
||||
_SETTINGS: Final = TypeAdapter(list[dict[str, JsonValue]])
|
||||
|
||||
|
||||
def _day(days_ago: int) -> str:
|
||||
return (datetime.now(timezone.utc) - timedelta(days=days_ago)).strftime("%Y-%m-%d")
|
||||
|
||||
|
||||
def _seed_daily_tag_spend(tag: str, days: tuple[str, ...]) -> None:
|
||||
with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection:
|
||||
for day in days:
|
||||
connection.execute(
|
||||
'INSERT INTO "LiteLLM_DailyTagSpend" (id, tag, date, api_key, model, spend, updated_at) '
|
||||
"VALUES (%s, %s, %s, %s, %s, 1.0, now())",
|
||||
(uuid.uuid4().hex, tag, day, f"integration-{tag}", "gpt-4o-mini"),
|
||||
)
|
||||
|
||||
|
||||
def _seed_old_spend_log(request_id: str, days_ago: int) -> None:
|
||||
with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection:
|
||||
connection.execute(
|
||||
'INSERT INTO "LiteLLM_SpendLogs" (request_id, call_type, api_key, spend, "startTime", "endTime") '
|
||||
"VALUES (%s, 'acompletion', %s, 0, now() - make_interval(days => %s), now() - make_interval(days => %s))",
|
||||
(request_id, f"integration-{request_id}", str(days_ago), str(days_ago)),
|
||||
)
|
||||
|
||||
|
||||
def _delete_daily_tag_spend(tag: str) -> None:
|
||||
with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection:
|
||||
connection.execute('DELETE FROM "LiteLLM_DailyTagSpend" WHERE tag = %s', (tag,))
|
||||
|
||||
|
||||
def _remaining_days(tag: str) -> tuple[str, ...]:
|
||||
rows: Final = read_rows('SELECT date FROM "LiteLLM_DailyTagSpend" WHERE tag = %s ORDER BY date', (tag,))
|
||||
return tuple(str(row["date"]) for row in rows)
|
||||
|
||||
|
||||
def _spend_log_present(request_id: str) -> bool:
|
||||
return bool(read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id = %s', (request_id,)))
|
||||
|
||||
|
||||
def _stored_retention_setting() -> JsonValue:
|
||||
rows: Final = read_rows(
|
||||
'SELECT param_value -> %s AS value FROM "LiteLLM_Config" WHERE param_name = %s',
|
||||
(RETENTION_SETTING, "general_settings"),
|
||||
)
|
||||
return rows[0]["value"] if rows else None
|
||||
|
||||
|
||||
def _store_retention_setting(value: JsonValue) -> None:
|
||||
with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection:
|
||||
if value is None:
|
||||
connection.execute(
|
||||
'UPDATE "LiteLLM_Config" SET param_value = param_value - %s WHERE param_name = %s',
|
||||
(RETENTION_SETTING, "general_settings"),
|
||||
)
|
||||
return
|
||||
connection.execute(
|
||||
'UPDATE "LiteLLM_Config" SET param_value = jsonb_set(param_value, ARRAY[%s], %s::jsonb) '
|
||||
"WHERE param_name = %s",
|
||||
(RETENTION_SETTING, json.dumps(value), "general_settings"),
|
||||
)
|
||||
|
||||
|
||||
def _listening_workers(owned: OwnedProxy) -> tuple[psutil.Process, ...]:
|
||||
port: Final = owned.gateway.client.base_url.port
|
||||
return tuple(
|
||||
child
|
||||
for child in psutil.Process(owned.process.pid).children(recursive=True)
|
||||
if any(conn.status == psutil.CONN_LISTEN and conn.laddr.port == port for conn in child.net_connections("inet"))
|
||||
)
|
||||
|
||||
|
||||
def _listed_retention_value(gateway: Gateway) -> JsonValue:
|
||||
listed: Final = _SETTINGS.validate_json(
|
||||
gateway.request("GET", "/config/list", params={"config_type": "general_settings"}).content
|
||||
)
|
||||
matching: Final = tuple(entry for entry in listed if entry["field_name"] == RETENTION_SETTING)
|
||||
return matching[0]["field_value"] if matching else "not listed"
|
||||
|
||||
|
||||
def _completion_id(gateway: Gateway, model: str) -> str:
|
||||
return string_value(gateway.chat(model, text=f"retention audit {uuid.uuid4().hex}")["id"])
|
||||
|
||||
|
||||
def _cleanup_config(tmp_path: Path, retention: dict[str, JsonValue]) -> Path:
|
||||
base: Final = _MAPPING.validate_python(yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()))
|
||||
config: Final = {
|
||||
**base,
|
||||
"general_settings": {
|
||||
**_MAPPING.validate_python(base["general_settings"]),
|
||||
**retention,
|
||||
"maximum_spend_logs_cleanup_cron": CLEANUP_EVERY_MINUTE,
|
||||
"scheduled_job_stagger": {"enabled": False},
|
||||
},
|
||||
}
|
||||
path: Final = tmp_path / "retention.yaml"
|
||||
path.write_text(yaml.safe_dump(config))
|
||||
return path
|
||||
|
||||
|
||||
@pytest.mark.timeout(240)
|
||||
def test_daily_tag_spend_retention_prunes_only_rows_older_than_the_period(gateway: Gateway, tmp_path: Path) -> None:
|
||||
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
expired, on_the_cutoff, today = _day(200), _day(30), _day(0)
|
||||
_seed_daily_tag_spend(tag, (expired, on_the_cutoff, today))
|
||||
try:
|
||||
config: Final = _cleanup_config(tmp_path, {"maximum_daily_tag_spend_retention_period": "30d"})
|
||||
with owned_proxy(gateway, tmp_path, {}, config=config):
|
||||
remaining: Final = eventually(
|
||||
lambda: _remaining_days(tag),
|
||||
lambda days: expired not in days,
|
||||
seconds=150,
|
||||
)
|
||||
assert remaining == (on_the_cutoff, today), remaining
|
||||
finally:
|
||||
_delete_daily_tag_spend(tag)
|
||||
|
||||
|
||||
@pytest.mark.timeout(240)
|
||||
def test_config_update_turns_on_daily_tag_spend_cleanup_without_a_restart(gateway: Gateway, tmp_path: Path) -> None:
|
||||
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
expired, yesterday_of_cutoff, on_the_cutoff, today = _day(200), _day(31), _day(30), _day(0)
|
||||
_seed_daily_tag_spend(tag, (expired, yesterday_of_cutoff, on_the_cutoff, today))
|
||||
previously_stored: Final = _stored_retention_setting()
|
||||
_store_retention_setting(None)
|
||||
try:
|
||||
config: Final = _cleanup_config(tmp_path, {})
|
||||
with owned_proxy(gateway, tmp_path, {}, config=config, workers=2) as owned, owned.scenario() as scenario:
|
||||
model: Final = scenario.model()
|
||||
assert _listed_retention_value(owned) is None
|
||||
owned.post("/config/update", {"general_settings": {RETENTION_SETTING: "30d"}})
|
||||
assert _listed_retention_value(owned) == "30d"
|
||||
remaining: Final = eventually(
|
||||
lambda: _remaining_days(tag),
|
||||
lambda days: yesterday_of_cutoff not in days,
|
||||
seconds=150,
|
||||
)
|
||||
assert remaining == (on_the_cutoff, today), remaining
|
||||
assert _completion_id(owned, model).startswith("chatcmpl-")
|
||||
finally:
|
||||
_store_retention_setting(previously_stored)
|
||||
_delete_daily_tag_spend(tag)
|
||||
|
||||
|
||||
@pytest.mark.timeout(240)
|
||||
def test_unparseable_daily_tag_spend_retention_deletes_nothing_and_keeps_serving(
|
||||
gateway: Gateway, tmp_path: Path
|
||||
) -> None:
|
||||
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
request_id: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
expired: Final = _day(200)
|
||||
_seed_daily_tag_spend(tag, (expired,))
|
||||
_seed_old_spend_log(request_id, days_ago=200)
|
||||
try:
|
||||
config: Final = _cleanup_config(
|
||||
tmp_path, {RETENTION_SETTING: "soon", "maximum_spend_logs_retention_period": "30d"}
|
||||
)
|
||||
with owned_proxy(gateway, tmp_path, {}, config=config) as owned, owned.scenario() as scenario:
|
||||
model: Final = scenario.model()
|
||||
eventually(lambda: _spend_log_present(request_id), lambda present: not present, seconds=150)
|
||||
assert _remaining_days(tag) == (expired,)
|
||||
assert _completion_id(owned, model).startswith("chatcmpl-")
|
||||
finally:
|
||||
_delete_daily_tag_spend(tag)
|
||||
|
||||
|
||||
@pytest.mark.timeout(240)
|
||||
def test_daily_tag_spend_keeps_days_the_shorter_spend_log_horizon_already_pruned(
|
||||
gateway: Gateway, tmp_path: Path
|
||||
) -> None:
|
||||
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
request_id: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
expired, inside_tag_horizon = _day(200), _day(60)
|
||||
_seed_daily_tag_spend(tag, (expired, inside_tag_horizon))
|
||||
_seed_old_spend_log(request_id, days_ago=60)
|
||||
try:
|
||||
config: Final = _cleanup_config(
|
||||
tmp_path, {RETENTION_SETTING: "90d", "maximum_spend_logs_retention_period": "30d"}
|
||||
)
|
||||
with owned_proxy(gateway, tmp_path, {}, config=config):
|
||||
eventually(lambda: _spend_log_present(request_id), lambda present: not present, seconds=150)
|
||||
remaining: Final = eventually(lambda: _remaining_days(tag), lambda days: expired not in days, seconds=150)
|
||||
assert remaining == (inside_tag_horizon,), remaining
|
||||
finally:
|
||||
_delete_daily_tag_spend(tag)
|
||||
|
||||
|
||||
@pytest.mark.timeout(240)
|
||||
def test_daily_tag_spend_cleanup_completes_after_one_of_two_workers_is_killed(gateway: Gateway, tmp_path: Path) -> None:
|
||||
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
expired, today = _day(200), _day(0)
|
||||
_seed_daily_tag_spend(tag, (expired, today))
|
||||
try:
|
||||
config: Final = _cleanup_config(tmp_path, {RETENTION_SETTING: "30d"})
|
||||
with owned_proxy_process(gateway, tmp_path, {}, config=config, workers=2) as owned:
|
||||
with owned.gateway.scenario() as scenario:
|
||||
model: Final = scenario.model()
|
||||
workers: Final = eventually(
|
||||
lambda: _listening_workers(owned), lambda found: len(found) == 2, seconds=30
|
||||
)
|
||||
workers[0].send_signal(signal.SIGKILL)
|
||||
eventually(lambda: workers[0].is_running(), lambda alive: not alive, seconds=10)
|
||||
ids: Final = tuple(_completion_id(owned.gateway, model) for _ in range(6))
|
||||
assert len(set(ids)) == 6 and all(identity.startswith("chatcmpl-") for identity in ids), ids
|
||||
remaining: Final = eventually(
|
||||
lambda: _remaining_days(tag), lambda days: expired not in days, seconds=150
|
||||
)
|
||||
assert remaining == (today,), remaining
|
||||
finally:
|
||||
_delete_daily_tag_spend(tag)
|
||||
|
||||
|
||||
@pytest.mark.timeout(240)
|
||||
def test_daily_tag_spend_is_kept_forever_when_its_retention_is_unset(gateway: Gateway, tmp_path: Path) -> None:
|
||||
tag: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
request_id: Final = f"integration-retention-{uuid.uuid4().hex}"
|
||||
expired: Final = _day(200)
|
||||
_seed_daily_tag_spend(tag, (expired,))
|
||||
_seed_old_spend_log(request_id, days_ago=200)
|
||||
try:
|
||||
config: Final = _cleanup_config(tmp_path, {"maximum_spend_logs_retention_period": "30d"})
|
||||
with owned_proxy(gateway, tmp_path, {}, config=config):
|
||||
eventually(lambda: _spend_log_present(request_id), lambda present: not present, seconds=150)
|
||||
assert _remaining_days(tag) == (expired,)
|
||||
finally:
|
||||
_delete_daily_tag_spend(tag)
|
||||
|
|
@ -133,46 +133,6 @@ def test_get_llm_provider_azure_o1():
|
|||
assert model == "o1-mini"
|
||||
|
||||
|
||||
def test_default_api_base():
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import (
|
||||
_get_openai_compatible_provider_info,
|
||||
)
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
# Patch environment variable to remove API base if it's set
|
||||
with patch.dict(os.environ, {}, clear=True):
|
||||
for provider in litellm.openai_compatible_providers:
|
||||
# Get the API base for the given provider
|
||||
if provider == "github_copilot":
|
||||
continue
|
||||
# Skip chatgpt as it requires OAuth authentication
|
||||
if provider == "chatgpt":
|
||||
continue
|
||||
# Skip ragflow as it requires specific model format: ragflow/chat/{id}/{model} or ragflow/agent/{id}/{model}
|
||||
if provider == "ragflow":
|
||||
continue
|
||||
_, _, _, api_base = _get_openai_compatible_provider_info(
|
||||
model=f"{provider}/*", api_base=None, api_key=None, dynamic_api_key=None
|
||||
)
|
||||
if api_base is None:
|
||||
continue
|
||||
|
||||
for other_provider in LlmProviders:
|
||||
if other_provider.value != provider and provider != "{}_chat".format(
|
||||
other_provider.value
|
||||
):
|
||||
if provider == "codestral" and other_provider.value == "mistral":
|
||||
continue
|
||||
elif provider == "github" and other_provider.value == "azure":
|
||||
continue
|
||||
elif (
|
||||
provider in ("qwencloud", "qwen_ai_platform")
|
||||
and other_provider.value == "dashscope"
|
||||
):
|
||||
continue
|
||||
assert other_provider.value not in api_base.replace("/openai", "")
|
||||
|
||||
|
||||
def test_hosted_vllm_default_api_key():
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import (
|
||||
_get_openai_compatible_provider_info,
|
||||
|
|
|
|||
|
|
@ -79,6 +79,7 @@ _PREVIOUSLY_DB_WINS: Final[tuple[str, ...]] = (
|
|||
"maximum_spend_logs_retention_period",
|
||||
"maximum_autorouter_session_retention_period",
|
||||
"maximum_health_check_retention_period",
|
||||
"maximum_daily_tag_spend_retention_period",
|
||||
"maximum_spend_logs_cleanup_batch_size",
|
||||
"maximum_spend_logs_cleanup_max_batches",
|
||||
"maximum_spend_logs_cleanup_run_budget",
|
||||
|
|
|
|||
|
|
@ -3862,6 +3862,7 @@ async def test_ProxyConfig__reschedule_spend_log_cleanup_job_health_check_retent
|
|||
async def test_ProxyConfig__update_general_settings_updates_health_check_retention(monkeypatch):
|
||||
settings = {}
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", settings)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", MagicMock(**{"get_job.return_value": None}))
|
||||
pc = ProxyConfig()
|
||||
reschedule = AsyncMock()
|
||||
monkeypatch.setattr(pc, "_reschedule_spend_log_cleanup_job", reschedule)
|
||||
|
|
@ -3872,6 +3873,329 @@ async def test_ProxyConfig__update_general_settings_updates_health_check_retenti
|
|||
reschedule.assert_awaited_once()
|
||||
|
||||
|
||||
def _paused_scheduler(monkeypatch):
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
real_scheduler = AsyncIOScheduler()
|
||||
real_scheduler.start(paused=True)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
|
||||
return real_scheduler
|
||||
|
||||
|
||||
def _scheduler_whose_first_add_job_raises(monkeypatch):
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
class FirstAddJobRaises(AsyncIOScheduler):
|
||||
raised = False
|
||||
|
||||
def add_job(self, *args, **kwargs):
|
||||
if not self.raised:
|
||||
self.raised = True
|
||||
raise RuntimeError("scheduler busy")
|
||||
return super().add_job(*args, **kwargs)
|
||||
|
||||
real_scheduler = FirstAddJobRaises()
|
||||
real_scheduler.start(paused=True)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
|
||||
return real_scheduler
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__reschedule_spend_log_cleanup_job_daily_tag_spend_retention(monkeypatch):
|
||||
real_scheduler = _paused_scheduler(monkeypatch)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.general_settings",
|
||||
{"maximum_daily_tag_spend_retention_period": "90d"},
|
||||
)
|
||||
pc = ProxyConfig()
|
||||
try:
|
||||
await pc._reschedule_spend_log_cleanup_job()
|
||||
job = real_scheduler.get_job("spend_log_cleanup_job")
|
||||
assert job is not None, "daily tag spend retention alone did not schedule the cleanup job"
|
||||
assert job.func.__name__ == "cleanup_old_spend_logs"
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_updates_daily_tag_spend_retention(monkeypatch):
|
||||
real_scheduler = _paused_scheduler(monkeypatch)
|
||||
pc = ProxyConfig()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
try:
|
||||
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
from litellm.proxy import proxy_server
|
||||
|
||||
assert proxy_server.general_settings["maximum_daily_tag_spend_retention_period"] == "90d"
|
||||
assert real_scheduler.get_job("spend_log_cleanup_job") is not None, "runtime retention did not schedule cleanup"
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_schedules_cleanup_when_db_row_was_already_applied(monkeypatch):
|
||||
"""A config reload applies the db row to the store before the side effects run, so the
|
||||
before/after snapshot is equal; the job must still be scheduled when none is running."""
|
||||
real_scheduler = _paused_scheduler(monkeypatch)
|
||||
pc = ProxyConfig()
|
||||
pc.settings.apply_db_row("general_settings", {"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
try:
|
||||
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
assert real_scheduler.get_job("spend_log_cleanup_job") is not None, "DB-only retention never scheduled cleanup"
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_retries_a_failed_schedule_once_per_settings_value(
|
||||
monkeypatch, caplog
|
||||
):
|
||||
"""An unparseable cron leaves no job behind; reloads must not retry it every tick, only when the
|
||||
cron or a retention value changes."""
|
||||
real_scheduler = _paused_scheduler(monkeypatch)
|
||||
pc = ProxyConfig()
|
||||
bad_cron = {"maximum_daily_tag_spend_retention_period": "90d", "maximum_spend_logs_cleanup_cron": "not a cron"}
|
||||
pc.settings.apply_db_row("general_settings", bad_cron)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
try:
|
||||
with caplog.at_level(logging.ERROR, logger="LiteLLM Proxy"):
|
||||
for _ in range(3):
|
||||
await pc._update_general_settings(bad_cron)
|
||||
assert real_scheduler.get_job("spend_log_cleanup_job") is None
|
||||
cron_errors = [r for r in caplog.records if "maximum_spend_logs_cleanup_cron" in r.getMessage()]
|
||||
assert len(cron_errors) == 1, f"invalid cron was retried on every reload: {len(cron_errors)} error lines"
|
||||
|
||||
await pc._update_general_settings({**bad_cron, "maximum_spend_logs_cleanup_cron": "* * * * *"})
|
||||
job = real_scheduler.get_job("spend_log_cleanup_job")
|
||||
assert job is not None, "a corrected cron did not schedule cleanup"
|
||||
assert "minute='*'" in str(job.trigger)
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_retries_a_schedule_that_raised(monkeypatch):
|
||||
"""A transient add_job failure must not be remembered as a completed attempt; the next
|
||||
reload with the same settings tries again."""
|
||||
real_scheduler = _scheduler_whose_first_add_job_raises(monkeypatch)
|
||||
pc = ProxyConfig()
|
||||
retention = {"maximum_daily_tag_spend_retention_period": "90d"}
|
||||
pc.settings.apply_db_row("general_settings", retention)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
try:
|
||||
await pc._update_general_settings(retention)
|
||||
assert real_scheduler.get_job("spend_log_cleanup_job") is None
|
||||
await pc._update_general_settings(retention)
|
||||
assert real_scheduler.get_job("spend_log_cleanup_job") is not None, "raised add_job was not retried"
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_retries_a_failed_replacement_of_the_live_job(monkeypatch):
|
||||
"""A cron change whose add_job raised keeps the old job running, so the next reload with the
|
||||
same settings must try the replacement again instead of leaving the new cron unapplied."""
|
||||
real_scheduler = _scheduler_whose_first_add_job_raises(monkeypatch)
|
||||
pc = ProxyConfig()
|
||||
pc.settings.load_yaml({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
real_scheduler.raised = True
|
||||
await pc._reschedule_spend_log_cleanup_job()
|
||||
real_scheduler.raised = False
|
||||
try:
|
||||
new_cron = {"maximum_spend_logs_cleanup_cron": "0 3 * * *"}
|
||||
await pc._update_general_settings(new_cron)
|
||||
assert "hour='3'" not in str(real_scheduler.get_job("spend_log_cleanup_job").trigger), "old job was lost"
|
||||
await pc._update_general_settings(new_cron)
|
||||
assert "hour='3'" in str(real_scheduler.get_job("spend_log_cleanup_job").trigger), (
|
||||
"failed replacement was not retried on the next sync"
|
||||
)
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_leaves_a_changed_db_schedule_to_startup_while_scheduler_is_stopped(
|
||||
monkeypatch,
|
||||
):
|
||||
"""The first DB sync runs before the scheduler starts and usually differs from the yaml; it
|
||||
must still leave registration to the startup block instead of adding a job it will replace."""
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
real_scheduler = AsyncIOScheduler()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
|
||||
pc = ProxyConfig()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
assert real_scheduler.get_jobs() == [], "DB sync registered the cleanup job before the scheduler started"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_leaves_first_registration_to_startup_while_scheduler_is_stopped(
|
||||
monkeypatch,
|
||||
):
|
||||
"""The DB sync that runs before the scheduler starts must not register the cleanup job; the
|
||||
startup block does, once, so the cross-replica stagger it applies to pending jobs survives."""
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
real_scheduler = AsyncIOScheduler()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
|
||||
pc = ProxyConfig()
|
||||
pc.settings.load_yaml({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
await pc._update_general_settings({"unrelated_key": "value"})
|
||||
assert real_scheduler.get_jobs() == [], "DB sync registered the cleanup job before the scheduler started"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_runtime_interval_job_carries_the_stagger_offset(monkeypatch):
|
||||
"""Once the scheduler is running the sync owns registration and the job it adds is staggered."""
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
from litellm.proxy.common_utils.scheduled_job_stagger import _OffsetTrigger
|
||||
|
||||
real_scheduler = AsyncIOScheduler()
|
||||
real_scheduler.start(paused=True)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
|
||||
pc = ProxyConfig()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
try:
|
||||
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
jobs = real_scheduler.get_jobs()
|
||||
assert [job.id for job in jobs] == ["spend_log_cleanup_job"]
|
||||
assert isinstance(jobs[0].trigger, _OffsetTrigger), repr(jobs[0].trigger)
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"bad_schedule",
|
||||
[
|
||||
{"maximum_spend_logs_cleanup_cron": "not a cron"},
|
||||
{"maximum_spend_logs_cleanup_cron": "0 0 * * * *"},
|
||||
{"maximum_spend_logs_retention_interval": "soon"},
|
||||
{"maximum_spend_logs_retention_interval": 86400},
|
||||
],
|
||||
)
|
||||
async def test_ProxyConfig__update_general_settings_keeps_the_live_cleanup_job_when_the_new_schedule_is_invalid(
|
||||
monkeypatch, bad_schedule
|
||||
):
|
||||
"""A schedule edit that does not parse must leave the old cleanup job running and must not
|
||||
stop the rest of the general settings sync."""
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
real_scheduler = AsyncIOScheduler()
|
||||
real_scheduler.start(paused=True)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
|
||||
ssrf_sync = MagicMock()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server._apply_ssrf_general_settings", ssrf_sync)
|
||||
pc = ProxyConfig()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
try:
|
||||
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
old_trigger = real_scheduler.get_job("spend_log_cleanup_job").trigger
|
||||
ssrf_sync.reset_mock()
|
||||
for _ in range(2):
|
||||
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d", **bad_schedule})
|
||||
live_job = real_scheduler.get_job("spend_log_cleanup_job")
|
||||
assert live_job is not None, "invalid schedule removed the cleanup job"
|
||||
assert live_job.trigger is old_trigger
|
||||
assert ssrf_sync.call_count == 2, "schedule error blocked the rest of the settings sync"
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_logs_an_overflowing_interval_once(monkeypatch, caplog):
|
||||
"""An interval that parses but overflows the trigger must keep the live job and log one
|
||||
error, not a traceback on every sync."""
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
real_scheduler = AsyncIOScheduler()
|
||||
real_scheduler.start(paused=True)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
|
||||
pc = ProxyConfig()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
try:
|
||||
await pc._update_general_settings({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
old_trigger = real_scheduler.get_job("spend_log_cleanup_job").trigger
|
||||
overflowing = {
|
||||
"maximum_daily_tag_spend_retention_period": "90d",
|
||||
"maximum_spend_logs_retention_interval": "99999999999d",
|
||||
}
|
||||
with caplog.at_level(logging.ERROR, logger="LiteLLM Proxy"):
|
||||
for _ in range(5):
|
||||
await pc._update_general_settings(overflowing)
|
||||
errors = [record for record in caplog.records if record.levelno >= logging.ERROR]
|
||||
assert len(errors) == 1, [record.getMessage() for record in errors]
|
||||
assert real_scheduler.get_job("spend_log_cleanup_job").trigger is old_trigger
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_reschedules_when_only_the_cron_changes(monkeypatch):
|
||||
real_scheduler = _paused_scheduler(monkeypatch)
|
||||
pc = ProxyConfig()
|
||||
pc.settings.load_yaml({"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
await pc._reschedule_spend_log_cleanup_job()
|
||||
try:
|
||||
interval_job = real_scheduler.get_job("spend_log_cleanup_job")
|
||||
assert interval_job is not None and "hour='3'" not in str(interval_job.trigger)
|
||||
|
||||
await pc._update_general_settings({"maximum_spend_logs_cleanup_cron": "0 3 * * *"})
|
||||
cron_job = real_scheduler.get_job("spend_log_cleanup_job")
|
||||
assert "hour='3'" in str(cron_job.trigger), "cron-only change did not reschedule"
|
||||
|
||||
await pc._update_general_settings({"maximum_spend_logs_cleanup_cron": "0 3 * * *"})
|
||||
assert real_scheduler.get_job("spend_log_cleanup_job") is cron_job, "unchanged cron replaced the job"
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ProxyConfig__update_general_settings_reschedules_a_cron_edit_the_reload_path_already_applied(
|
||||
monkeypatch,
|
||||
):
|
||||
"""The periodic reload applies the DB row through _update_config_from_db before
|
||||
_update_general_settings snapshots the previous schedule, so a cron edited in the DB must
|
||||
still replace the live job's trigger."""
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
real_scheduler = AsyncIOScheduler()
|
||||
real_scheduler.start(paused=True)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.scheduler", real_scheduler)
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
|
||||
pc = ProxyConfig()
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.general_settings", pc.settings)
|
||||
try:
|
||||
first_row = {"maximum_daily_tag_spend_retention_period": "90d", "maximum_spend_logs_cleanup_cron": "0 3 * * *"}
|
||||
pc.settings.apply_db_row("general_settings", first_row)
|
||||
await pc._update_general_settings(first_row)
|
||||
assert "hour='3'" in str(real_scheduler.get_job("spend_log_cleanup_job").trigger)
|
||||
|
||||
edited_row = {**first_row, "maximum_spend_logs_cleanup_cron": "0 5 * * *"}
|
||||
pc.settings.apply_db_row("general_settings", edited_row)
|
||||
await pc._update_general_settings(edited_row)
|
||||
assert "hour='5'" in str(real_scheduler.get_job("spend_log_cleanup_job").trigger), "DB cron edit was ignored"
|
||||
|
||||
pc.settings.apply_db_row("general_settings", edited_row)
|
||||
await pc._update_general_settings(edited_row)
|
||||
assert "hour='5'" in str(real_scheduler.get_job("spend_log_cleanup_job").trigger)
|
||||
finally:
|
||||
real_scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ProxyConfig._update_general_settings
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
@ -4003,6 +4327,7 @@ async def test_ProxyConfig__update_general_settings_skips_redundant_retention_re
|
|||
pc = ProxyConfig()
|
||||
reschedule: Final = AsyncMock()
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {})
|
||||
monkeypatch.setattr(proxy_server, "scheduler", MagicMock())
|
||||
monkeypatch.setattr(pc, "_reschedule_spend_log_cleanup_job", reschedule)
|
||||
|
||||
await pc._update_general_settings({"maximum_health_check_retention_period": "30d"})
|
||||
|
|
@ -4021,6 +4346,7 @@ async def test_ProxyConfig__update_general_settings_reschedules_after_retention_
|
|||
pc = ProxyConfig()
|
||||
reschedule: Final = AsyncMock()
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {})
|
||||
monkeypatch.setattr(proxy_server, "scheduler", MagicMock(**{"get_job.return_value": None}))
|
||||
monkeypatch.setattr(pc, "_reschedule_spend_log_cleanup_job", reschedule)
|
||||
|
||||
await pc._update_general_settings({"maximum_health_check_retention_period": "30d"})
|
||||
|
|
@ -4052,7 +4378,7 @@ async def test_ProxyConfig__update_general_settings_dispatches_every_side_effect
|
|||
if name == "_apply_cache_size_setting":
|
||||
handler.assert_awaited_once_with({}, cache_size_was_db=False)
|
||||
elif name == "_apply_retention_settings":
|
||||
handler.assert_awaited_once_with({}, previous_retention_values=())
|
||||
handler.assert_awaited_once_with({}, previous_cleanup_schedule=())
|
||||
elif name == "_apply_pass_through_settings":
|
||||
handler.assert_awaited_once_with({}, previous_endpoints=None)
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -935,6 +935,89 @@ async def test_periodic_reload_job_scheduled_without_store_model_in_db(monkeypat
|
|||
scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_initialize_scheduled_jobs_registers_cleanup_when_retention_lives_only_in_the_db(monkeypatch):
|
||||
"""With no config file, the startup DB sync rebinds general_settings to a store holding the
|
||||
retention period; the cleanup job must be registered from that live value, not the stale
|
||||
empty dict the caller passed in."""
|
||||
monkeypatch.delenv("DISABLE_PRISMA_SCHEMA_UPDATE", raising=False)
|
||||
monkeypatch.delenv("STORE_MODEL_IN_DB", raising=False)
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
from litellm.proxy.proxy_server import ProxyStartupEvent
|
||||
from litellm.proxy.utils import ProxyLogging
|
||||
|
||||
mock_prisma_client = MagicMock()
|
||||
mock_prisma_client.db.litellm_config.find_first = AsyncMock(return_value=None)
|
||||
mock_proxy_logging = MagicMock(spec=ProxyLogging)
|
||||
mock_proxy_logging.slack_alerting_instance = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_config = _mock_scheduled_proxy_config()
|
||||
db_settings = proxy_server_module.ProxyConfig().settings
|
||||
db_settings.apply_db_row("general_settings", {"maximum_daily_tag_spend_retention_period": "30d"})
|
||||
|
||||
async def sync_from_db(*args: object, **kwargs: object) -> None:
|
||||
proxy_server_module._bind_general_settings_store(db_settings)
|
||||
|
||||
mock_proxy_config.add_deployment.side_effect = sync_from_db
|
||||
scheduler = AsyncIOScheduler()
|
||||
try:
|
||||
with (
|
||||
patch("litellm.proxy.proxy_server.proxy_config", mock_proxy_config),
|
||||
patch("litellm.proxy.proxy_server.store_model_in_db", True),
|
||||
patch("litellm.proxy.proxy_server.general_settings", {}),
|
||||
patch("litellm.proxy.proxy_server.AsyncIOScheduler", return_value=scheduler),
|
||||
):
|
||||
await ProxyStartupEvent.initialize_scheduled_background_jobs(
|
||||
general_settings={},
|
||||
prisma_client=mock_prisma_client,
|
||||
proxy_budget_rescheduler_min_time=1,
|
||||
proxy_budget_rescheduler_max_time=2,
|
||||
proxy_batch_write_at=5,
|
||||
proxy_logging_obj=mock_proxy_logging,
|
||||
)
|
||||
assert scheduler.get_job("spend_log_cleanup_job") is not None, "DB-only retention was not scheduled at boot"
|
||||
finally:
|
||||
scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_initialize_scheduled_jobs_does_not_fall_back_to_the_interval_for_a_non_string_cron(monkeypatch):
|
||||
"""A truthy non-string cron is invalid, so startup must log it and register no cleanup job
|
||||
rather than silently pruning on the default interval the admin never configured."""
|
||||
monkeypatch.delenv("DISABLE_PRISMA_SCHEMA_UPDATE", raising=False)
|
||||
monkeypatch.delenv("STORE_MODEL_IN_DB", raising=False)
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
from litellm.proxy.proxy_server import ProxyStartupEvent
|
||||
from litellm.proxy.utils import ProxyLogging
|
||||
|
||||
mock_prisma_client = MagicMock()
|
||||
mock_proxy_logging = MagicMock(spec=ProxyLogging)
|
||||
mock_proxy_logging.slack_alerting_instance = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
settings = {"maximum_daily_tag_spend_retention_period": "30d", "maximum_spend_logs_cleanup_cron": 5}
|
||||
scheduler = AsyncIOScheduler()
|
||||
try:
|
||||
with (
|
||||
patch("litellm.proxy.proxy_server.proxy_config", _mock_scheduled_proxy_config()),
|
||||
patch("litellm.proxy.proxy_server.store_model_in_db", False),
|
||||
patch("litellm.proxy.proxy_server.general_settings", settings),
|
||||
patch("litellm.proxy.proxy_server.AsyncIOScheduler", return_value=scheduler),
|
||||
):
|
||||
await ProxyStartupEvent.initialize_scheduled_background_jobs(
|
||||
general_settings=settings,
|
||||
prisma_client=mock_prisma_client,
|
||||
proxy_budget_rescheduler_min_time=1,
|
||||
proxy_budget_rescheduler_max_time=2,
|
||||
proxy_batch_write_at=5,
|
||||
proxy_logging_obj=mock_proxy_logging,
|
||||
)
|
||||
assert scheduler.get_job("spend_log_cleanup_job") is None, "invalid cron fell back to the interval"
|
||||
finally:
|
||||
scheduler.shutdown(wait=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_initialize_scheduled_jobs_uses_configured_config_reload_interval(monkeypatch):
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -827,6 +827,29 @@ async def test_health_check_retention_alone_cleans_only_the_health_check_table()
|
|||
assert abs((cutoff_date - expected_cutoff).total_seconds()) < 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_daily_tag_spend_retention_alone_prunes_only_that_table_by_calendar_day():
|
||||
client = _mock_prisma_for_retention([0])
|
||||
cleaner = SpendLogCleanup(general_settings={"maximum_daily_tag_spend_retention_period": "90d"})
|
||||
cleaner.pod_lock_manager = None
|
||||
await cleaner.cleanup_old_spend_logs(client)
|
||||
tables = [call[0][0] for call in client.db.execute_raw.call_args_list]
|
||||
assert len(tables) == 1
|
||||
assert '"LiteLLM_DailyTagSpend"' in tables[0]
|
||||
cutoff_day = client.db.execute_raw.call_args[0][1]
|
||||
assert cutoff_day == (datetime.now(timezone.utc) - timedelta(days=90)).date().isoformat()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_spend_logs_retention_alone_keeps_daily_tag_spend_forever():
|
||||
client = _mock_prisma_for_retention([0, 0])
|
||||
cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
|
||||
cleaner.pod_lock_manager = None
|
||||
await cleaner.cleanup_old_spend_logs(client)
|
||||
tables = [call[0][0] for call in client.db.execute_raw.call_args_list]
|
||||
assert not any('"LiteLLM_DailyTagSpend"' in sql for sql in tables)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_each_retention_key_cuts_off_at_its_own_horizon():
|
||||
client = _mock_prisma_for_retention([0, 0, 0, 0, 0])
|
||||
|
|
|
|||
|
|
@ -58,6 +58,26 @@ def test_image_edit_forwards_provider_params_and_extra_body():
|
|||
assert response.data
|
||||
|
||||
|
||||
def test_image_edit_keeps_an_internal_prefixed_kwarg_out_of_the_provider_request():
|
||||
captured = {}
|
||||
client = HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(_capture_image_edit_request(captured))))
|
||||
|
||||
litellm.image_edit(
|
||||
model="openai/gpt-image-1",
|
||||
image=PNG_BYTES,
|
||||
prompt="add a hat",
|
||||
api_key="sk-test",
|
||||
api_base="https://edit.example/v1",
|
||||
client=client,
|
||||
seed=42,
|
||||
_litellm_undeclared_sentinel="internal",
|
||||
)
|
||||
|
||||
fields = _multipart_text_fields(captured["content_type"], captured["body"])
|
||||
assert "_litellm_undeclared_sentinel" not in fields
|
||||
assert fields["seed"] == "42"
|
||||
|
||||
|
||||
def test_image_edit_extra_body_takes_precedence_over_kwargs():
|
||||
captured = {}
|
||||
client = HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(_capture_image_edit_request(captured))))
|
||||
|
|
|
|||
29
tests/unit/images/test_main.py
Normal file
29
tests/unit/images/test_main.py
Normal file
|
|
@ -0,0 +1,29 @@
|
|||
import json
|
||||
from typing import Final
|
||||
|
||||
import httpx
|
||||
import respx
|
||||
|
||||
import litellm
|
||||
|
||||
|
||||
def test_image_generation_keeps_an_internal_prefixed_kwarg_out_of_the_provider_request(
|
||||
respx_mock: respx.MockRouter,
|
||||
) -> None:
|
||||
api_base: Final = "http://localhost:12346/v1"
|
||||
mock_route: Final = respx_mock.post(url__regex=rf"{api_base}/images/generations.*").mock(
|
||||
return_value=httpx.Response(status_code=200, json={"created": 1712697600, "data": [{"b64_json": "aW1n"}]})
|
||||
)
|
||||
|
||||
litellm.image_generation(
|
||||
model="openai/gpt-image-1",
|
||||
prompt="a red circle",
|
||||
api_base=api_base,
|
||||
api_key="fake_openai_api_key",
|
||||
_litellm_undeclared_sentinel="internal",
|
||||
)
|
||||
|
||||
assert mock_route.called
|
||||
sent: Final = json.loads(respx_mock.calls[0].request.content)
|
||||
assert "_litellm_undeclared_sentinel" not in sent, sent
|
||||
assert sent["prompt"] == "a red circle"
|
||||
|
|
@ -84,6 +84,29 @@ class TestBedrockFilesTransformation:
|
|||
"max_tokens" in model_input
|
||||
), f"Record {i+1} should have max_tokens"
|
||||
|
||||
def test_batch_keeps_an_internal_prefixed_key_out_of_the_bedrock_model_input(self):
|
||||
from litellm.llms.bedrock.files.transformation import BedrockFilesConfig
|
||||
|
||||
result: Final = BedrockFilesConfig()._transform_openai_jsonl_content_to_bedrock_jsonl_content(
|
||||
[
|
||||
{
|
||||
"custom_id": "internal-key-1",
|
||||
"method": "POST",
|
||||
"url": "/v1/chat/completions",
|
||||
"body": {
|
||||
"model": "anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"max_tokens": 10,
|
||||
"_litellm_undeclared_sentinel": "internal",
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
model_input: Final = json.dumps(result[0]["modelInput"])
|
||||
assert "_litellm_undeclared_sentinel" not in model_input, model_input
|
||||
assert result[0]["modelInput"]["max_tokens"] == 10
|
||||
|
||||
def test_nova_text_only_uses_converse_format(self):
|
||||
"""
|
||||
Test that Nova models produce Converse API format in batch modelInput.
|
||||
|
|
|
|||
|
|
@ -1,5 +1,11 @@
|
|||
import pytest
|
||||
import json
|
||||
from typing import Final
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
import respx
|
||||
|
||||
import litellm
|
||||
from litellm.llms.elevenlabs.text_to_speech.transformation import (
|
||||
ElevenLabsTextToSpeechConfig,
|
||||
)
|
||||
|
|
@ -16,10 +22,7 @@ def test_should_encode_elevenlabs_voice_id_path_segment():
|
|||
},
|
||||
)
|
||||
|
||||
assert (
|
||||
url
|
||||
== "https://api.elevenlabs.io/v1/text-to-speech/voice%2F..%2F..%2Fmodels%3Fx%3D1%23frag"
|
||||
)
|
||||
assert url == "https://api.elevenlabs.io/v1/text-to-speech/voice%2F..%2F..%2Fmodels%3Fx%3D1%23frag"
|
||||
|
||||
|
||||
def test_should_reject_dot_segment_elevenlabs_voice_id():
|
||||
|
|
@ -31,3 +34,24 @@ def test_should_reject_dot_segment_elevenlabs_voice_id():
|
|||
api_base="https://api.elevenlabs.io",
|
||||
litellm_params={config.ELEVENLABS_VOICE_ID_KEY: ".."},
|
||||
)
|
||||
|
||||
|
||||
def test_speech_keeps_an_internal_prefixed_kwarg_out_of_the_elevenlabs_request(respx_mock: respx.MockRouter) -> None:
|
||||
api_base: Final = "http://localhost:12346"
|
||||
mock_route: Final = respx_mock.post(url__regex=rf"{api_base}/v1/text-to-speech/.*").mock(
|
||||
return_value=httpx.Response(status_code=200, content=b"audio", headers={"content-type": "audio/mpeg"})
|
||||
)
|
||||
|
||||
litellm.speech(
|
||||
model="elevenlabs/eleven_multilingual_v2",
|
||||
input="hi",
|
||||
voice="21m00Tcm4TlvDq8ikWAM",
|
||||
api_base=api_base,
|
||||
api_key="fake_elevenlabs_api_key",
|
||||
_litellm_undeclared_sentinel="internal",
|
||||
)
|
||||
|
||||
assert mock_route.called
|
||||
sent: Final = json.loads(respx_mock.calls[0].request.content)
|
||||
assert "_litellm_undeclared_sentinel" not in sent, sent
|
||||
assert sent["text"] == "hi"
|
||||
|
|
|
|||
|
|
@ -395,6 +395,34 @@ def test_completion_strips_eager_input_streaming_before_openai(respx_mock: respx
|
|||
assert sent_tool["function"]["name"] == "write_file"
|
||||
|
||||
|
||||
def test_embedding_keeps_an_internal_prefixed_kwarg_out_of_the_provider_request(respx_mock: respx.MockRouter) -> None:
|
||||
api_base: Final = "http://localhost:12346/v1"
|
||||
mock_route: Final = respx_mock.post(url__regex=rf"{api_base}/embeddings.*").mock(
|
||||
return_value=httpx.Response(
|
||||
status_code=200,
|
||||
json={
|
||||
"object": "list",
|
||||
"data": [{"object": "embedding", "index": 0, "embedding": [0.1, 0.2]}],
|
||||
"model": "text-embedding-3-small",
|
||||
"usage": {"prompt_tokens": 1, "total_tokens": 1},
|
||||
},
|
||||
)
|
||||
)
|
||||
|
||||
litellm.embedding(
|
||||
model="openai/text-embedding-3-small",
|
||||
input="hi",
|
||||
api_base=api_base,
|
||||
api_key="fake_openai_api_key",
|
||||
_litellm_undeclared_sentinel="internal",
|
||||
)
|
||||
|
||||
assert mock_route.called
|
||||
sent: Final = json.loads(respx_mock.calls[0].request.content)
|
||||
assert "_litellm_undeclared_sentinel" not in sent, sent
|
||||
assert sent["model"] == "text-embedding-3-small"
|
||||
|
||||
|
||||
def test_custom_provider_with_extra_headers():
|
||||
|
||||
with patch.object(
|
||||
|
|
|
|||
|
|
@ -262,7 +262,7 @@ OWNED_NAMES: Final = (
|
|||
*PRICING_NAMES,
|
||||
)
|
||||
|
||||
Classifier: TypeAlias = Callable[[dict[str, object]], dict[str, object]] # mutable-ok: classifiers use dict
|
||||
Classifier: TypeAlias = Callable[[Mapping[str, object]], Mapping[str, object]]
|
||||
|
||||
CLASSIFIERS: Final[Mapping[str, Classifier]] = MappingProxyType(
|
||||
{ # pyright: ignore[reportUnknownArgumentType] # untyped legacy classifiers
|
||||
|
|
@ -279,18 +279,31 @@ def test_owned_name_is_kept_out_of_provider_params(name: str, classifier_name: s
|
|||
provider_value: Final = object()
|
||||
classify: Final = CLASSIFIERS[classifier_name]
|
||||
|
||||
result: Final = classify({name: object(), PROVIDER_KNOB: provider_value}) # mutable-ok: classifiers take a dict
|
||||
result: Final = classify(MappingProxyType({name: object(), PROVIDER_KNOB: provider_value}))
|
||||
|
||||
assert result == MappingProxyType({PROVIDER_KNOB: provider_value})
|
||||
assert result[PROVIDER_KNOB] is provider_value
|
||||
|
||||
|
||||
def test_a_name_no_object_declares_reaches_the_provider() -> None:
|
||||
result: Final = CLASSIFIERS["completion"]({PROVIDER_KNOB: 1}) # mutable-ok: classifier input type
|
||||
result: Final = CLASSIFIERS["completion"](MappingProxyType({PROVIDER_KNOB: 1}))
|
||||
|
||||
assert result == MappingProxyType({PROVIDER_KNOB: 1})
|
||||
|
||||
|
||||
@pytest.mark.parametrize("classifier_name", CLASSIFIERS)
|
||||
def test_an_undeclared_internal_prefixed_name_is_kept_out_of_provider_params(classifier_name: str) -> None:
|
||||
undeclared: Final = "_litellm_never_declared_anywhere"
|
||||
lookalike: Final = "provider_litellm_knob"
|
||||
assert undeclared not in all_litellm_params
|
||||
|
||||
result: Final = CLASSIFIERS[classifier_name](
|
||||
MappingProxyType({undeclared: object(), PROVIDER_KNOB: 1, lookalike: 2})
|
||||
)
|
||||
|
||||
assert result == MappingProxyType({PROVIDER_KNOB: 1, lookalike: 2})
|
||||
|
||||
|
||||
def _cache_key_for_model_group(cache: Cache, model_group: str, options: CachingOptions) -> str:
|
||||
return cache.get_cache_key( # pyright: ignore[reportUnknownMemberType] # untyped legacy key builder
|
||||
model=model_group,
|
||||
|
|
@ -421,9 +434,7 @@ CARRIED_PARAMS: Final = tuple(
|
|||
def test_every_param_get_litellm_params_carries_is_kept_out_of_provider_params(name: str) -> None:
|
||||
provider_value: Final = object()
|
||||
|
||||
result: Final = CLASSIFIERS["completion"](
|
||||
{name: object(), PROVIDER_KNOB: provider_value} # mutable-ok: classifier input type
|
||||
)
|
||||
result: Final = CLASSIFIERS["completion"](MappingProxyType({name: object(), PROVIDER_KNOB: provider_value}))
|
||||
|
||||
assert result == MappingProxyType({PROVIDER_KNOB: provider_value})
|
||||
|
||||
|
|
|
|||
5
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
5
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -28325,6 +28325,11 @@ export interface components {
|
|||
* @description Maximum retention period for auto-router benchmark session rollup rows (e.g., '365d'). Rows whose last turn is older than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rollup rows are never deleted.
|
||||
*/
|
||||
maximum_autorouter_session_retention_period?: string | null;
|
||||
/**
|
||||
* Maximum Daily Tag Spend Retention Period
|
||||
* @description Maximum retention period for per-day tag spend aggregate rows (e.g., '90d'). Rows whose day is older than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rows are never deleted. Only historical tag usage analytics are affected; tag budgets read the lifetime counter.
|
||||
*/
|
||||
maximum_daily_tag_spend_retention_period?: string | null;
|
||||
/**
|
||||
* Maximum Health Check Retention Period
|
||||
* @description Maximum retention period for health-check rows (e.g., '30d'). Rows whose checked_at is older than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rows are never deleted. Set this well above health_check_interval because /health and the UI read the latest row per model.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue