diff --git a/docs/my-website/docs/proxy/configs.md b/docs/my-website/docs/proxy/configs.md index 28b0b67e336..1adc4943d33 100644 --- a/docs/my-website/docs/proxy/configs.md +++ b/docs/my-website/docs/proxy/configs.md @@ -692,9 +692,13 @@ general_settings: allowed_routes: ["route1", "route2"] # list of allowed proxy API routes - a user can access. (currently JWT-Auth only) key_management_system: google_kms # either google_kms or azure_kms master_key: string + + # Database Settings database_url: string database_connection_pool_limit: 0 # default 100 database_connection_timeout: 0 # default 60s + allow_requests_on_db_unavailable: boolean # if true, will allow requests that can not connect to the DB to verify Virtual Key to still work + custom_auth: string max_parallel_requests: 0 # the max parallel requests allowed per deployment global_max_parallel_requests: 0 # the max parallel requests allowed on the proxy all up @@ -766,6 +770,7 @@ general_settings: | database_url | string | The URL for the database connection [Set up Virtual Keys](virtual_keys) | | database_connection_pool_limit | integer | The limit for database connection pool [Setting DB Connection Pool limit](#configure-db-pool-limits--connection-timeouts) | | database_connection_timeout | integer | The timeout for database connections in seconds [Setting DB Connection Pool limit, timeout](#configure-db-pool-limits--connection-timeouts) | +| allow_requests_on_db_unavailable | boolean | If true, allows requests to succeed even if DB is unreachable. **Only use this if running LiteLLM in your VPC** This will allow requests to work even when LiteLLM cannot connect to the DB to verify a Virtual Key | | custom_auth | string | Write your own custom authentication logic [Doc Custom Auth](virtual_keys#custom-auth) | | max_parallel_requests | integer | The max parallel requests allowed per deployment | | global_max_parallel_requests | integer | The max parallel requests allowed on the proxy overall | diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index e00d494d94b..d2b04122b43 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -13,6 +13,7 @@ import traceback from datetime import datetime from typing import TYPE_CHECKING, Any, List, Literal, Optional +import httpx from pydantic import BaseModel import litellm @@ -717,12 +718,49 @@ async def get_key_object( ) return _response - except Exception: + except httpx.ConnectError as e: + return await _handle_failed_db_connection_for_get_key_object(e=e) + except Exception as e: raise Exception( f"Key doesn't exist in db. key={hashed_token}. Create key via `/key/generate` call." ) +async def _handle_failed_db_connection_for_get_key_object( + e: Exception, +) -> UserAPIKeyAuth: + """ + Handles httpx.ConnectError when reading a Virtual Key from LiteLLM DB + + Use this if you don't want failed DB queries to block LLM API reqiests + + Returns: + - UserAPIKeyAuth: If general_settings.allow_requests_on_db_unavailable is True + + Raises: + - Orignal Exception in all other cases + """ + from litellm.proxy.proxy_server import general_settings, proxy_logging_obj + + # If this flag is on, requests failing to connect to the DB will be allowed + if general_settings.get("allow_requests_on_db_unavailable", False) is True: + # log to prometheus + proxy_logging_obj.service_logging_obj.service_success_hook( + service=ServiceTypes.ALLOW_REQUESTS_ON_DB_UNAVAILABLE, + call_type="get_key_object", + parent_otel_span=None, + duration=0.0, + start_time=None, + end_time=None, + ) + + return UserAPIKeyAuth( + key_name="failed-to-connect-to-db", token="failed-to-connect-to-db" + ) + else: + raise e + + @log_to_opentelemetry async def get_org_object( org_id: str, diff --git a/litellm/proxy/proxy_config.yaml b/litellm/proxy/proxy_config.yaml index acae35530bd..53ab370010e 100644 --- a/litellm/proxy/proxy_config.yaml +++ b/litellm/proxy/proxy_config.yaml @@ -6,22 +6,10 @@ model_list: api_base: https://exampleopenaiendpoint-production.up.railway.app/ +litellm_settings: + callbacks: ["prometheus"] + service_callback: ["prometheus_system"] + general_settings: - disable_prisma_schema_update: true - # Temoporarily disable callbacks and key and budget constraints for large batch inference jobs - # alerting: ["slack"] - alerting_threshold: 300 # sends alerts if requests hang for 5min+ and responses take 5min+ -litellm_settings: # module level litellm settings - https://github.com/BerriAI/litellm/blob/main/litellm/__init__.py - success_callback: ["prometheus"] - service_callback: ["prometheus_system"] - # callbacks: ["otel"] - # upperbound_key_generate_params: - # max_budget: 5000 # upperbound of $5000, for all /key/generate requests - # duration: "7d" # upperbound of 7 days for all /key/generate requests - drop_params: True # Raise an exception if the openai param being passed in isn't supported. - # set_verbose: True - json_logs: true - cache: false -router_settings: - routing_strategy: simple-shuffle # "simple-shuffle" shown to result in highest throughput. https://docs.litellm.ai/docs/proxy/configs#load-balancing \ No newline at end of file + allow_requests_on_db_unavailable: true diff --git a/litellm/types/services.py b/litellm/types/services.py index cfa427ebc3d..6ffdfc8916f 100644 --- a/litellm/types/services.py +++ b/litellm/types/services.py @@ -12,6 +12,7 @@ class ServiceTypes(str, enum.Enum): REDIS = "redis" DB = "postgres" + ALLOW_REQUESTS_ON_DB_UNAVAILABLE = "allow_requests_on_db_unavailable" BATCH_WRITE_TO_DB = "batch_write_to_db" LITELLM = "self" ROUTER = "router"