Merge branch 'litellm_allow_running_in_vpc_mode' into litellm_dev_nov6_e

This commit is contained in:
Ishaan Jaff 2024-11-06 06:23:47 -08:00
commit 3b37ea5d1f
4 changed files with 52 additions and 4 deletions

View file

@ -692,9 +692,13 @@ general_settings:
allowed_routes: ["route1", "route2"] # list of allowed proxy API routes - a user can access. (currently JWT-Auth only)
key_management_system: google_kms # either google_kms or azure_kms
master_key: string
# Database Settings
database_url: string
database_connection_pool_limit: 0 # default 100
database_connection_timeout: 0 # default 60s
allow_requests_on_db_unavailable: boolean # if true, will allow requests that can not connect to the DB to verify Virtual Key to still work
custom_auth: string
max_parallel_requests: 0 # the max parallel requests allowed per deployment
global_max_parallel_requests: 0 # the max parallel requests allowed on the proxy all up
@ -766,6 +770,7 @@ general_settings:
| database_url | string | The URL for the database connection [Set up Virtual Keys](virtual_keys) |
| database_connection_pool_limit | integer | The limit for database connection pool [Setting DB Connection Pool limit](#configure-db-pool-limits--connection-timeouts) |
| database_connection_timeout | integer | The timeout for database connections in seconds [Setting DB Connection Pool limit, timeout](#configure-db-pool-limits--connection-timeouts) |
| allow_requests_on_db_unavailable | boolean | If true, allows requests to succeed even if DB is unreachable. **Only use this if running LiteLLM in your VPC** This will allow requests to work even when LiteLLM cannot connect to the DB to verify a Virtual Key |
| custom_auth | string | Write your own custom authentication logic [Doc Custom Auth](virtual_keys#custom-auth) |
| max_parallel_requests | integer | The max parallel requests allowed per deployment |
| global_max_parallel_requests | integer | The max parallel requests allowed on the proxy overall |

View file

@ -13,6 +13,7 @@ import traceback
from datetime import datetime
from typing import TYPE_CHECKING, Any, List, Literal, Optional
import httpx
from pydantic import BaseModel
import litellm
@ -717,12 +718,49 @@ async def get_key_object(
)
return _response
except Exception:
except httpx.ConnectError as e:
return await _handle_failed_db_connection_for_get_key_object(e=e)
except Exception as e:
raise Exception(
f"Key doesn't exist in db. key={hashed_token}. Create key via `/key/generate` call."
)
async def _handle_failed_db_connection_for_get_key_object(
e: Exception,
) -> UserAPIKeyAuth:
"""
Handles httpx.ConnectError when reading a Virtual Key from LiteLLM DB
Use this if you don't want failed DB queries to block LLM API reqiests
Returns:
- UserAPIKeyAuth: If general_settings.allow_requests_on_db_unavailable is True
Raises:
- Orignal Exception in all other cases
"""
from litellm.proxy.proxy_server import general_settings, proxy_logging_obj
# If this flag is on, requests failing to connect to the DB will be allowed
if general_settings.get("allow_requests_on_db_unavailable", False) is True:
# log to prometheus
proxy_logging_obj.service_logging_obj.service_success_hook(
service=ServiceTypes.ALLOW_REQUESTS_ON_DB_UNAVAILABLE,
call_type="get_key_object",
parent_otel_span=None,
duration=0.0,
start_time=None,
end_time=None,
)
return UserAPIKeyAuth(
key_name="failed-to-connect-to-db", token="failed-to-connect-to-db"
)
else:
raise e
@log_to_opentelemetry
async def get_org_object(
org_id: str,

View file

@ -6,6 +6,10 @@ model_list:
api_base: https://exampleopenaiendpoint-production.up.railway.app/
general_settings:
alerting: ["slack"]
alerting_threshold: 0.001
litellm_settings:
callbacks: ["prometheus"]
service_callback: ["prometheus_system"]
general_settings:
allow_requests_on_db_unavailable: true

View file

@ -12,6 +12,7 @@ class ServiceTypes(str, enum.Enum):
REDIS = "redis"
DB = "postgres"
ALLOW_REQUESTS_ON_DB_UNAVAILABLE = "allow_requests_on_db_unavailable"
BATCH_WRITE_TO_DB = "batch_write_to_db"
LITELLM = "self"
ROUTER = "router"