mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
update code structure, move hard coded values to const and make the reslve function readable by moving fallback logic to a seperate function
This commit is contained in:
parent
2c14d421dc
commit
6a3ca9a629
2 changed files with 43 additions and 22 deletions
|
|
@ -420,6 +420,7 @@ DEFAULT_MAX_TOKENS_FOR_TRITON = int(os.getenv("DEFAULT_MAX_TOKENS_FOR_TRITON", 2
|
|||
# per-request/model timeout—see ``CompletionTimeout.resolve`` in completion_timeout.py. MCP uses
|
||||
# dedicated timeouts (e.g. `MCP_CLIENT_TIMEOUT`), not `request_timeout`.
|
||||
DEFAULT_REQUEST_TIMEOUT_SECONDS: float = 6000.0
|
||||
COMPLETION_HTTP_FALLBACK_SECONDS: float = 600.0
|
||||
request_timeout: float = float(
|
||||
os.getenv("REQUEST_TIMEOUT", str(int(DEFAULT_REQUEST_TIMEOUT_SECONDS)))
|
||||
)
|
||||
|
|
|
|||
|
|
@ -6,58 +6,78 @@ from typing import Callable, Optional, Union
|
|||
|
||||
import httpx
|
||||
|
||||
from litellm.constants import DEFAULT_REQUEST_TIMEOUT_SECONDS
|
||||
from litellm.constants import (
|
||||
COMPLETION_HTTP_FALLBACK_SECONDS,
|
||||
DEFAULT_REQUEST_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
class CompletionTimeout:
|
||||
"""Resolves HTTP timeout for ``completion()`` from model vs global settings."""
|
||||
|
||||
@staticmethod
|
||||
def _fallback_when_no_explicit_timeout(
|
||||
global_timeout: Optional[Union[float, str]],
|
||||
) -> float:
|
||||
"""
|
||||
Used when ``model_timeout`` and kwargs timeouts are all unset.
|
||||
|
||||
``global_timeout`` is :attr:`litellm.request_timeout` (numeric / string), not
|
||||
:class:`httpx.Timeout`.
|
||||
|
||||
If it equals :data:`~litellm.constants.DEFAULT_REQUEST_TIMEOUT_SECONDS` (6000),
|
||||
return :data:`~litellm.constants.COMPLETION_HTTP_FALLBACK_SECONDS`. Same if
|
||||
``None``. Otherwise return ``float(global_timeout)``.
|
||||
"""
|
||||
if global_timeout is None:
|
||||
return COMPLETION_HTTP_FALLBACK_SECONDS
|
||||
if float(global_timeout) == float(DEFAULT_REQUEST_TIMEOUT_SECONDS):
|
||||
return COMPLETION_HTTP_FALLBACK_SECONDS
|
||||
return float(global_timeout)
|
||||
|
||||
@staticmethod
|
||||
def resolve(
|
||||
model_timeout: Optional[Union[float, str, httpx.Timeout]],
|
||||
kwargs: dict,
|
||||
custom_llm_provider: str,
|
||||
*,
|
||||
global_timeout: Optional[Union[float, str, httpx.Timeout]],
|
||||
global_timeout: Optional[Union[float, str]],
|
||||
supports_httpx_timeout: Callable[[str], bool],
|
||||
) -> Union[float, httpx.Timeout]:
|
||||
"""
|
||||
Order: ``model_timeout`` (call argument / merged ``litellm_params``), then
|
||||
``kwargs["timeout"]``, ``kwargs["request_timeout"]``, then ``global_timeout``
|
||||
(e.g. :attr:`litellm.request_timeout` from proxy ``litellm_settings``), else ``600``.
|
||||
Resolution order (first non-None wins):
|
||||
|
||||
Coerce :class:`httpx.Timeout` when the provider does not support it. If the value
|
||||
came only from ``global_timeout`` (no model/kwargs timeout) and equals ``6000``
|
||||
(:data:`~litellm.constants.DEFAULT_REQUEST_TIMEOUT_SECONDS`), use ``600`` for
|
||||
completion so chat calls do not inherit the long package default; explicit
|
||||
``model_timeout`` / kwargs values of ``6000`` are left unchanged.
|
||||
1. ``model_timeout`` (call argument / merged ``litellm_params``)
|
||||
2. ``kwargs["timeout"]``
|
||||
3. ``kwargs["request_timeout"]``
|
||||
4. Fallback from ``global_timeout`` (:attr:`litellm.request_timeout`) — if it is
|
||||
the package default (6000), use 600 instead.
|
||||
|
||||
Coerce :class:`httpx.Timeout` when the provider does not support it.
|
||||
Explicit ``6000`` on the model or in kwargs is kept as ``6000``.
|
||||
"""
|
||||
timeout_from_global_only = False
|
||||
resolved: Union[float, str, httpx.Timeout]
|
||||
if model_timeout is not None:
|
||||
resolved: Union[float, str, httpx.Timeout] = model_timeout
|
||||
resolved = model_timeout
|
||||
elif kwargs.get("timeout") is not None:
|
||||
resolved = kwargs["timeout"]
|
||||
elif kwargs.get("request_timeout") is not None:
|
||||
resolved = kwargs["request_timeout"]
|
||||
else:
|
||||
resolved = global_timeout if global_timeout is not None else 600
|
||||
timeout_from_global_only = True
|
||||
resolved = CompletionTimeout._fallback_when_no_explicit_timeout(
|
||||
global_timeout
|
||||
)
|
||||
|
||||
if isinstance(resolved, httpx.Timeout) and not supports_httpx_timeout(
|
||||
custom_llm_provider
|
||||
):
|
||||
read_timeout = resolved.read
|
||||
resolved = (
|
||||
float(read_timeout) if read_timeout is not None else 600.0
|
||||
float(read_timeout)
|
||||
if read_timeout is not None
|
||||
else COMPLETION_HTTP_FALLBACK_SECONDS
|
||||
) # default 10 min timeout
|
||||
elif not isinstance(resolved, httpx.Timeout):
|
||||
resolved = float(resolved) # type: ignore
|
||||
|
||||
if (
|
||||
timeout_from_global_only
|
||||
and not isinstance(resolved, httpx.Timeout)
|
||||
and float(resolved) == float(DEFAULT_REQUEST_TIMEOUT_SECONDS)
|
||||
):
|
||||
resolved = 600.0
|
||||
|
||||
return resolved
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue