From 6a3ca9a62909fccd8a29a75f5fdde7024a275c64 Mon Sep 17 00:00:00 2001 From: harish876 Date: Tue, 14 Apr 2026 23:06:30 +0000 Subject: [PATCH] update code structure, move hard coded values to const and make the reslve function readable by moving fallback logic to a seperate function --- litellm/constants.py | 1 + .../litellm_core_utils/completion_timeout.py | 64 ++++++++++++------- 2 files changed, 43 insertions(+), 22 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index 1703102bc84..a25ecc17cd1 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -420,6 +420,7 @@ DEFAULT_MAX_TOKENS_FOR_TRITON = int(os.getenv("DEFAULT_MAX_TOKENS_FOR_TRITON", 2 # per-request/model timeout—see ``CompletionTimeout.resolve`` in completion_timeout.py. MCP uses # dedicated timeouts (e.g. `MCP_CLIENT_TIMEOUT`), not `request_timeout`. DEFAULT_REQUEST_TIMEOUT_SECONDS: float = 6000.0 +COMPLETION_HTTP_FALLBACK_SECONDS: float = 600.0 request_timeout: float = float( os.getenv("REQUEST_TIMEOUT", str(int(DEFAULT_REQUEST_TIMEOUT_SECONDS))) ) diff --git a/litellm/litellm_core_utils/completion_timeout.py b/litellm/litellm_core_utils/completion_timeout.py index 73451fb3f86..5350d88e593 100644 --- a/litellm/litellm_core_utils/completion_timeout.py +++ b/litellm/litellm_core_utils/completion_timeout.py @@ -6,58 +6,78 @@ from typing import Callable, Optional, Union import httpx -from litellm.constants import DEFAULT_REQUEST_TIMEOUT_SECONDS +from litellm.constants import ( + COMPLETION_HTTP_FALLBACK_SECONDS, + DEFAULT_REQUEST_TIMEOUT_SECONDS, +) class CompletionTimeout: """Resolves HTTP timeout for ``completion()`` from model vs global settings.""" + @staticmethod + def _fallback_when_no_explicit_timeout( + global_timeout: Optional[Union[float, str]], + ) -> float: + """ + Used when ``model_timeout`` and kwargs timeouts are all unset. + + ``global_timeout`` is :attr:`litellm.request_timeout` (numeric / string), not + :class:`httpx.Timeout`. + + If it equals :data:`~litellm.constants.DEFAULT_REQUEST_TIMEOUT_SECONDS` (6000), + return :data:`~litellm.constants.COMPLETION_HTTP_FALLBACK_SECONDS`. Same if + ``None``. Otherwise return ``float(global_timeout)``. + """ + if global_timeout is None: + return COMPLETION_HTTP_FALLBACK_SECONDS + if float(global_timeout) == float(DEFAULT_REQUEST_TIMEOUT_SECONDS): + return COMPLETION_HTTP_FALLBACK_SECONDS + return float(global_timeout) + @staticmethod def resolve( model_timeout: Optional[Union[float, str, httpx.Timeout]], kwargs: dict, custom_llm_provider: str, *, - global_timeout: Optional[Union[float, str, httpx.Timeout]], + global_timeout: Optional[Union[float, str]], supports_httpx_timeout: Callable[[str], bool], ) -> Union[float, httpx.Timeout]: """ - Order: ``model_timeout`` (call argument / merged ``litellm_params``), then - ``kwargs["timeout"]``, ``kwargs["request_timeout"]``, then ``global_timeout`` - (e.g. :attr:`litellm.request_timeout` from proxy ``litellm_settings``), else ``600``. + Resolution order (first non-None wins): - Coerce :class:`httpx.Timeout` when the provider does not support it. If the value - came only from ``global_timeout`` (no model/kwargs timeout) and equals ``6000`` - (:data:`~litellm.constants.DEFAULT_REQUEST_TIMEOUT_SECONDS`), use ``600`` for - completion so chat calls do not inherit the long package default; explicit - ``model_timeout`` / kwargs values of ``6000`` are left unchanged. + 1. ``model_timeout`` (call argument / merged ``litellm_params``) + 2. ``kwargs["timeout"]`` + 3. ``kwargs["request_timeout"]`` + 4. Fallback from ``global_timeout`` (:attr:`litellm.request_timeout`) — if it is + the package default (6000), use 600 instead. + + Coerce :class:`httpx.Timeout` when the provider does not support it. + Explicit ``6000`` on the model or in kwargs is kept as ``6000``. """ - timeout_from_global_only = False + resolved: Union[float, str, httpx.Timeout] if model_timeout is not None: - resolved: Union[float, str, httpx.Timeout] = model_timeout + resolved = model_timeout elif kwargs.get("timeout") is not None: resolved = kwargs["timeout"] elif kwargs.get("request_timeout") is not None: resolved = kwargs["request_timeout"] else: - resolved = global_timeout if global_timeout is not None else 600 - timeout_from_global_only = True + resolved = CompletionTimeout._fallback_when_no_explicit_timeout( + global_timeout + ) if isinstance(resolved, httpx.Timeout) and not supports_httpx_timeout( custom_llm_provider ): read_timeout = resolved.read resolved = ( - float(read_timeout) if read_timeout is not None else 600.0 + float(read_timeout) + if read_timeout is not None + else COMPLETION_HTTP_FALLBACK_SECONDS ) # default 10 min timeout elif not isinstance(resolved, httpx.Timeout): resolved = float(resolved) # type: ignore - if ( - timeout_from_global_only - and not isinstance(resolved, httpx.Timeout) - and float(resolved) == float(DEFAULT_REQUEST_TIMEOUT_SECONDS) - ): - resolved = 600.0 - return resolved