mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
fix request timeout if using default value from constants.py
This commit is contained in:
parent
de20182232
commit
c2b26ed562
2 changed files with 31 additions and 4 deletions
|
|
@ -393,7 +393,16 @@ MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = int(
|
|||
)
|
||||
DEFAULT_MAX_TOKENS_FOR_TRITON = int(os.getenv("DEFAULT_MAX_TOKENS_FOR_TRITON", 2000))
|
||||
#### Networking settings ####
|
||||
request_timeout: float = float(os.getenv("REQUEST_TIMEOUT", 600)) # time in seconds
|
||||
# Sentinel used when `REQUEST_TIMEOUT` is unset: `litellm.request_timeout` keeps this
|
||||
# value so longer-running surfaces (Router `timeout or litellm.request_timeout`,
|
||||
# speech/TTS, responses, vector stores, etc.) get a long HTTP deadline. Chat
|
||||
# `completion()` maps this sentinel down to 600s when the caller did not set a
|
||||
# per-request/model timeout—see `_resolve_completion_timeout` in main.py. MCP uses
|
||||
# dedicated timeouts (e.g. `MCP_CLIENT_TIMEOUT`), not `request_timeout`.
|
||||
DEFAULT_REQUEST_TIMEOUT_SECONDS: float = 6000.0
|
||||
request_timeout: float = float(
|
||||
os.getenv("REQUEST_TIMEOUT", str(int(DEFAULT_REQUEST_TIMEOUT_SECONDS)))
|
||||
)
|
||||
DEFAULT_A2A_AGENT_TIMEOUT: float = float(
|
||||
os.getenv("DEFAULT_A2A_AGENT_TIMEOUT", 6000)
|
||||
) # 10 minutes
|
||||
|
|
|
|||
|
|
@ -68,6 +68,7 @@ if TYPE_CHECKING:
|
|||
from litellm.constants import (
|
||||
DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT,
|
||||
DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT,
|
||||
DEFAULT_REQUEST_TIMEOUT_SECONDS,
|
||||
)
|
||||
from litellm.exceptions import LiteLLMUnknownProvider
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
|
|
@ -1060,9 +1061,14 @@ def _resolve_completion_timeout(
|
|||
- **Model config alias:** ``kwargs["request_timeout"]`` when the caller passes the
|
||||
per-model ``request_timeout`` field from model config (same idea as deployment
|
||||
`litellm_params.request_timeout`).
|
||||
- **Global proxy settings:** :attr:`litellm.request_timeout`, set from
|
||||
``litellm_settings.request_timeout`` when the proxy loads config.
|
||||
- **Default:** ``600`` seconds if nothing above is set.
|
||||
- **Global module default:** :attr:`litellm.request_timeout` (from
|
||||
``litellm_settings.request_timeout`` on the proxy when set, otherwise
|
||||
:data:`~litellm.constants.DEFAULT_REQUEST_TIMEOUT_SECONDS`, i.e. ``6000`` seconds).
|
||||
That long default is shared with Router, speech/TTS, and other subsystems; for
|
||||
chat completion only, if the timeout came solely from this module attribute and
|
||||
still equals ``6000``, it is treated as unset and ``600`` seconds is used instead.
|
||||
|
||||
- **Fallback:** ``600`` seconds if no timeout is resolved above.
|
||||
|
||||
Also accepts ``kwargs["timeout"]`` as a fallback when the named ``timeout`` argument
|
||||
is omitted.
|
||||
|
|
@ -1076,10 +1082,22 @@ def _resolve_completion_timeout(
|
|||
timeout = kwargs.get("timeout")
|
||||
if timeout is None:
|
||||
timeout = kwargs.get("request_timeout")
|
||||
resolved_from_litellm_request_timeout_attr = False
|
||||
if timeout is None:
|
||||
timeout = getattr(litellm, "request_timeout", None)
|
||||
if timeout is not None:
|
||||
resolved_from_litellm_request_timeout_attr = True
|
||||
if timeout is None:
|
||||
timeout = 600
|
||||
elif (
|
||||
resolved_from_litellm_request_timeout_attr
|
||||
and not isinstance(timeout, httpx.Timeout)
|
||||
and float(timeout) == float(DEFAULT_REQUEST_TIMEOUT_SECONDS)
|
||||
):
|
||||
# 6000s is the package default for litellm.request_timeout so MCP, speech/TTS,
|
||||
# Router, and similar paths keep a long deadline. completion() uses 600s when
|
||||
# nothing more specific was supplied (explicit kwargs still win above).
|
||||
timeout = 600
|
||||
if isinstance(timeout, httpx.Timeout) and not supports_httpx_timeout(
|
||||
custom_llm_provider
|
||||
):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue