diff --git a/litellm/__init__.py b/litellm/__init__.py index 91457f9b049..84fc7fc6789 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -7,6 +7,7 @@ import threading import os from typing import Callable, List, Optional, Dict, Union, Any, Literal, get_args from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler +from litellm import constants from litellm.caching.caching import Cache, DualCache, RedisCache, InMemoryCache from litellm.types.llms.bedrock import COHERE_EMBEDDING_INPUT_TYPES from litellm.types.utils import ( diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 421a588e390..015292db183 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -1972,6 +1972,13 @@ class ProxyConfig: raise Exception( f"Invalid value set for upperbound_key_generate_params - value={value}" ) + elif key == "constants": + # allow overriding constants + for constant_key, constant_value in value.items(): + verbose_proxy_logger.debug( + f"{blue_color_code} setting litellm.constants.{constant_key}={constant_value}{reset_color_code}" + ) + setattr(litellm.constants, constant_key, constant_value) else: verbose_proxy_logger.debug( f"{blue_color_code} setting litellm.{key}={value}{reset_color_code}" @@ -8850,3 +8857,9 @@ app.include_router(openai_files_router) app.include_router(team_callback_router) app.include_router(budget_management_router) app.include_router(model_management_router) + +app.include_router(ui_crud_endpoints_router) +app.include_router(openai_files_router) +app.include_router(team_callback_router) +app.include_router(budget_management_router) +app.include_router(model_management_router) diff --git a/litellm/router.py b/litellm/router.py index 722eed56f1e..0dd8e2ec928 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -68,7 +68,6 @@ from litellm.router_utils.batch_utils import ( from litellm.router_utils.client_initalization_utils import InitalizeOpenAISDKClient from litellm.router_utils.cooldown_cache import CooldownCache from litellm.router_utils.cooldown_handlers import ( - DEFAULT_COOLDOWN_TIME_SECONDS, _async_get_cooldown_deployments, _async_get_cooldown_deployments_with_debug_info, _get_cooldown_deployments, @@ -383,7 +382,9 @@ class Router: self.allowed_fails = allowed_fails else: self.allowed_fails = litellm.allowed_fails - self.cooldown_time = cooldown_time or DEFAULT_COOLDOWN_TIME_SECONDS + self.cooldown_time = ( + cooldown_time or litellm.constants.DEFAULT_COOLDOWN_TIME_SECONDS + ) self.cooldown_cache = CooldownCache( cache=self.cache, default_cooldown_time=self.cooldown_time ) diff --git a/litellm/router_utils/cooldown_handlers.py b/litellm/router_utils/cooldown_handlers.py index 8f5c3895a6d..4cbb3895af3 100644 --- a/litellm/router_utils/cooldown_handlers.py +++ b/litellm/router_utils/cooldown_handlers.py @@ -11,11 +11,6 @@ from typing import TYPE_CHECKING, Any, List, Optional, Union import litellm from litellm._logging import verbose_router_logger -from litellm.constants import ( - DEFAULT_COOLDOWN_TIME_SECONDS, - DEFAULT_FAILURE_THRESHOLD_PERCENT, - SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD, -) from litellm.router_utils.cooldown_callbacks import router_cooldown_event_callback from .router_callbacks.track_deployment_metrics import ( @@ -197,12 +192,12 @@ def _should_cooldown_deployment( elif ( percent_fails == 1.0 and total_requests_this_minute - >= SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD + >= litellm.constants.SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD ): # Cooldown if all requests failed and we have reasonable traffic return True elif ( - percent_fails > DEFAULT_FAILURE_THRESHOLD_PERCENT + percent_fails > litellm.constants.DEFAULT_FAILURE_THRESHOLD_PERCENT and not is_single_deployment_model_group # by default we should avoid cooldowns on single deployment model groups ): return True @@ -377,7 +372,8 @@ def should_cooldown_based_on_allowed_fails_policy( or litellm_router_instance.allowed_fails ) cooldown_time = ( - litellm_router_instance.cooldown_time or DEFAULT_COOLDOWN_TIME_SECONDS + litellm_router_instance.cooldown_time + or litellm.constants.DEFAULT_COOLDOWN_TIME_SECONDS ) current_fails = litellm_router_instance.failed_calls.get_cache(key=deployment) or 0