diff --git a/litellm/llms/custom_httpx/http_handler.py b/litellm/llms/custom_httpx/http_handler.py index 3a67d5127b7..303b02a7497 100644 --- a/litellm/llms/custom_httpx/http_handler.py +++ b/litellm/llms/custom_httpx/http_handler.py @@ -657,7 +657,13 @@ class AsyncHTTPHandler: ) return LiteLLMAiohttpTransport( client=lambda: ClientSession( - connector=TCPConnector(**connector_kwargs), + connector=TCPConnector( + limit=0, # 0 = unlimited connections per host + keepalive_timeout=120, # Keep connections alive for 2 minutes (default is 15s) + ttl_dns_cache=300, # Cache DNS for 5 minutes + enable_cleanup_closed=True, + **connector_kwargs + ), trust_env=trust_env, ), ) diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index ae449bbb15b..bda2049ddec 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -240,6 +240,17 @@ class BaseLLMHTTPHandler: signed_json_body: Optional[bytes] = None, shared_session: Optional["ClientSession"] = None, ): + # PANIC: Ensure shared_session is being passed for connection reuse + from litellm._logging import verbose_logger + if shared_session is None and client is None: + error_msg = ( + "❌ PANIC: shared_session is None in async_completion! " + "This means connection reuse is broken. Session should be passed from completion() -> async_completion(). " + f"Provider: {custom_llm_provider}, Model: {model}" + ) + verbose_logger.error(error_msg) + raise ValueError(error_msg) + if client is None: verbose_logger.debug( f"Creating HTTP client with shared_session: {id(shared_session) if shared_session else None}" @@ -426,6 +437,7 @@ class BaseLLMHTTPHandler: ), json_mode=json_mode, signed_json_body=signed_json_body, + shared_session=shared_session, ) if stream is True: diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 5409d73164b..8652a99fc17 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -44,6 +44,7 @@ from litellm.types.utils import ( from litellm.utils import load_credentials_from_list if TYPE_CHECKING: + from aiohttp import ClientSession from opentelemetry.trace import Span as _Span from litellm.integrations.opentelemetry import OpenTelemetry @@ -559,7 +560,7 @@ async def proxy_shutdown_event(): @asynccontextmanager async def proxy_startup_event(app: FastAPI): - global prisma_client, master_key, use_background_health_checks, llm_router, llm_model_list, general_settings, proxy_budget_rescheduler_min_time, proxy_budget_rescheduler_max_time, litellm_proxy_admin_name, db_writer_client, store_model_in_db, premium_user, _license_check, proxy_batch_polling_interval + global prisma_client, master_key, use_background_health_checks, llm_router, llm_model_list, general_settings, proxy_budget_rescheduler_min_time, proxy_budget_rescheduler_max_time, litellm_proxy_admin_name, db_writer_client, store_model_in_db, premium_user, _license_check, proxy_batch_polling_interval, shared_aiohttp_session import json init_verbose_loggers() @@ -674,10 +675,39 @@ async def proxy_startup_event(app: FastAPI): ## [Optional] Initialize dd tracer ProxyStartupEvent._init_dd_tracer() + ## Initialize shared aiohttp session for connection reuse + try: + from aiohttp import ClientSession, TCPConnector + + # Create connector with connection pooling settings optimized for long-lived connections + connector = TCPConnector( + limit=100, # Max 100 connections per host + keepalive_timeout=120, # Keep connections alive for 2 minutes (prevents "once in a while" cold starts) + ttl_dns_cache=300, # Cache DNS for 5 minutes + enable_cleanup_closed=True, + force_close=False, # Don't force close connections after each request + ) + + shared_aiohttp_session = ClientSession(connector=connector) + verbose_proxy_logger.info( + f"🔄 SESSION REUSE: Created shared aiohttp session for connection pooling (ID: {id(shared_aiohttp_session)})" + ) + except Exception as e: + verbose_proxy_logger.warning( + f"⚠️ Failed to create shared aiohttp session: {e}. Continuing without session reuse." + ) + # End of startup event yield - # Shutdown event + # Shutdown event - close shared aiohttp session + if shared_aiohttp_session is not None: + try: + await shared_aiohttp_session.close() + verbose_proxy_logger.info("🔄 SESSION REUSE: Closed shared aiohttp session") + except Exception as e: + verbose_proxy_logger.error(f"Error closing shared aiohttp session: {e}") + await proxy_shutdown_event() @@ -955,6 +985,7 @@ worker_config = None master_key: Optional[str] = None otel_logging = False prisma_client: Optional[PrismaClient] = None +shared_aiohttp_session: Optional["ClientSession"] = None # Global shared session for connection reuse user_api_key_cache = DualCache( default_in_memory_ttl=UserAPIKeyCacheTTLEnum.in_memory_cache_ttl.value ) diff --git a/litellm/proxy/route_llm_request.py b/litellm/proxy/route_llm_request.py index e1a6ca2a2be..8d271d45498 100644 --- a/litellm/proxy/route_llm_request.py +++ b/litellm/proxy/route_llm_request.py @@ -86,6 +86,15 @@ async def route_request( """ Common helper to route the request """ + # Add shared aiohttp session for connection reuse (prevents cold starts) + try: + from litellm.proxy.proxy_server import shared_aiohttp_session + if shared_aiohttp_session is not None and not shared_aiohttp_session.closed: + data["shared_session"] = shared_aiohttp_session + except Exception: + # Silently continue without session reuse if import fails or session unavailable + pass + team_id = get_team_id_from_data(data) router_model_names = llm_router.model_names if llm_router is not None else []