From c2b26ed56223b029ebb5c7a09b1ed7ca791d797f Mon Sep 17 00:00:00 2001 From: harish876 Date: Tue, 14 Apr 2026 00:50:43 +0000 Subject: [PATCH] fix request timeout if using default value from constants.py --- litellm/constants.py | 11 ++++++++++- litellm/main.py | 24 +++++++++++++++++++++--- 2 files changed, 31 insertions(+), 4 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index b1b6e2cf09b..0cf9d98b2aa 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -393,7 +393,16 @@ MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = int( ) DEFAULT_MAX_TOKENS_FOR_TRITON = int(os.getenv("DEFAULT_MAX_TOKENS_FOR_TRITON", 2000)) #### Networking settings #### -request_timeout: float = float(os.getenv("REQUEST_TIMEOUT", 600)) # time in seconds +# Sentinel used when `REQUEST_TIMEOUT` is unset: `litellm.request_timeout` keeps this +# value so longer-running surfaces (Router `timeout or litellm.request_timeout`, +# speech/TTS, responses, vector stores, etc.) get a long HTTP deadline. Chat +# `completion()` maps this sentinel down to 600s when the caller did not set a +# per-request/model timeout—see `_resolve_completion_timeout` in main.py. MCP uses +# dedicated timeouts (e.g. `MCP_CLIENT_TIMEOUT`), not `request_timeout`. +DEFAULT_REQUEST_TIMEOUT_SECONDS: float = 6000.0 +request_timeout: float = float( + os.getenv("REQUEST_TIMEOUT", str(int(DEFAULT_REQUEST_TIMEOUT_SECONDS))) +) DEFAULT_A2A_AGENT_TIMEOUT: float = float( os.getenv("DEFAULT_A2A_AGENT_TIMEOUT", 6000) ) # 10 minutes diff --git a/litellm/main.py b/litellm/main.py index 8e4872931f7..bb467002ec7 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -68,6 +68,7 @@ if TYPE_CHECKING: from litellm.constants import ( DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT, DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT, + DEFAULT_REQUEST_TIMEOUT_SECONDS, ) from litellm.exceptions import LiteLLMUnknownProvider from litellm.integrations.custom_logger import CustomLogger @@ -1060,9 +1061,14 @@ def _resolve_completion_timeout( - **Model config alias:** ``kwargs["request_timeout"]`` when the caller passes the per-model ``request_timeout`` field from model config (same idea as deployment `litellm_params.request_timeout`). - - **Global proxy settings:** :attr:`litellm.request_timeout`, set from - ``litellm_settings.request_timeout`` when the proxy loads config. - - **Default:** ``600`` seconds if nothing above is set. + - **Global module default:** :attr:`litellm.request_timeout` (from + ``litellm_settings.request_timeout`` on the proxy when set, otherwise + :data:`~litellm.constants.DEFAULT_REQUEST_TIMEOUT_SECONDS`, i.e. ``6000`` seconds). + That long default is shared with Router, speech/TTS, and other subsystems; for + chat completion only, if the timeout came solely from this module attribute and + still equals ``6000``, it is treated as unset and ``600`` seconds is used instead. + + - **Fallback:** ``600`` seconds if no timeout is resolved above. Also accepts ``kwargs["timeout"]`` as a fallback when the named ``timeout`` argument is omitted. @@ -1076,10 +1082,22 @@ def _resolve_completion_timeout( timeout = kwargs.get("timeout") if timeout is None: timeout = kwargs.get("request_timeout") + resolved_from_litellm_request_timeout_attr = False if timeout is None: timeout = getattr(litellm, "request_timeout", None) + if timeout is not None: + resolved_from_litellm_request_timeout_attr = True if timeout is None: timeout = 600 + elif ( + resolved_from_litellm_request_timeout_attr + and not isinstance(timeout, httpx.Timeout) + and float(timeout) == float(DEFAULT_REQUEST_TIMEOUT_SECONDS) + ): + # 6000s is the package default for litellm.request_timeout so MCP, speech/TTS, + # Router, and similar paths keep a long deadline. completion() uses 600s when + # nothing more specific was supplied (explicit kwargs still win above). + timeout = 600 if isinstance(timeout, httpx.Timeout) and not supports_httpx_timeout( custom_llm_provider ):