mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(lens): use async-timeout on python 3.10 for budget reservation timeouts (#44911)
* fix(lens): use async-timeout on python 3.10 for budget reservation timeouts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(deps): keep uv.lock diff to the async-timeout entry Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(lens): cover real request deadline expiry in reserved_budget Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(lens): finish Python 3.10 timeout coverage and dependency checks * test(lens): control event-loop time for deadline regressions * test(tracing): include priced call count in trace fixture * fix(lens): limit timeout compatibility changes to PR scope --------- Co-authored-by: Moe Khalil <moe@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
a3e74a5223
commit
acd2a7ccc5
6 changed files with 131 additions and 7 deletions
7
.github/workflows/_test-unit-base.yml
vendored
7
.github/workflows/_test-unit-base.yml
vendored
|
|
@ -3,6 +3,11 @@ name: _Unit Test Base (Reusable)
|
|||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
python-version:
|
||||
description: "Python version used to install dependencies and run tests"
|
||||
required: false
|
||||
type: string
|
||||
default: "3.12"
|
||||
test-path:
|
||||
description: >-
|
||||
Space-separated pytest paths to run. A path that no longer exists is
|
||||
|
|
@ -78,7 +83,7 @@ permissions:
|
|||
contents: read
|
||||
|
||||
env:
|
||||
UV_PYTHON: "3.12"
|
||||
UV_PYTHON: ${{ inputs.python-version }}
|
||||
LITELLM_LOCAL_MODEL_COST_MAP: "True"
|
||||
|
||||
jobs:
|
||||
|
|
|
|||
15
.github/workflows/test-unit-proxy-db.yml
vendored
15
.github/workflows/test-unit-proxy-db.yml
vendored
|
|
@ -39,6 +39,21 @@ concurrency:
|
|||
# pinning the whole file to one worker (the default --dist=loadscope
|
||||
# behavior for single-file targets).
|
||||
jobs:
|
||||
lens-python-310:
|
||||
name: Lens Python 3.10
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
pull-requests: write
|
||||
uses: ./.github/workflows/_test-unit-base.yml
|
||||
with:
|
||||
python-version: "3.10"
|
||||
test-path: tests/unit/proxy/lens/test_inference.py
|
||||
workers: 0
|
||||
reruns: 0
|
||||
timeout-minutes: 5
|
||||
artifact-name: lens-python-310
|
||||
|
||||
# Fast guard — fails the workflow when a test directory or file inside a sharded
|
||||
# tree is claimed by no shard. The semantic-shard design has no catch-all bucket,
|
||||
# so an unassigned child runs nowhere; assert_ci_coverage.py holds the tree list
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import asyncio
|
||||
import sys
|
||||
from collections.abc import AsyncGenerator, Callable, Coroutine
|
||||
from contextlib import asynccontextmanager
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
|
@ -25,6 +26,11 @@ from litellm.types.llms.base import LiteLLMBaseModel
|
|||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.utils import CostPerToken, ModelResponse
|
||||
|
||||
if sys.version_info >= (3, 11):
|
||||
from asyncio import timeout
|
||||
else:
|
||||
from async_timeout import timeout
|
||||
|
||||
BUDGET_LEASE: Final = timedelta(minutes=5)
|
||||
BUDGET_RENEW_INTERVAL: Final = 30.0
|
||||
BUDGET_WAIT_TIMEOUT: Final = 60.0
|
||||
|
|
@ -317,7 +323,7 @@ async def renew_budget_reservation(
|
|||
while True:
|
||||
await asyncio.sleep(BUDGET_RENEW_INTERVAL)
|
||||
try:
|
||||
async with asyncio.timeout(BUDGET_RENEW_INTERVAL):
|
||||
async with timeout(BUDGET_RENEW_INTERVAL):
|
||||
if (
|
||||
await repo.update(
|
||||
lens_id, lambda e: renew_reservation(e, reservation_id, datetime.now(timezone.utc))
|
||||
|
|
@ -325,7 +331,7 @@ async def renew_budget_reservation(
|
|||
is None
|
||||
):
|
||||
raise HTTPException(503, "Could not renew analysis budget reservation")
|
||||
except TimeoutError as error:
|
||||
except (TimeoutError, asyncio.TimeoutError) as error:
|
||||
raise HTTPException(503, "Analysis budget reservation renewal timed out") from error
|
||||
|
||||
|
||||
|
|
@ -363,15 +369,15 @@ async def reserved_budget(
|
|||
repo: LensRepository, lens_id: str, reservation_id: str, reserve: Callable[[Lens], Lens], admitted: asyncio.Event
|
||||
) -> AsyncGenerator[None]:
|
||||
try:
|
||||
async with asyncio.timeout(float(litellm.request_timeout)):
|
||||
async with timeout(float(litellm.request_timeout)):
|
||||
try:
|
||||
async with asyncio.timeout(BUDGET_WAIT_TIMEOUT):
|
||||
async with timeout(BUDGET_WAIT_TIMEOUT):
|
||||
await wait_for_reservation(repo, lens_id, reservation_id, reserve)
|
||||
except TimeoutError as error:
|
||||
except (TimeoutError, asyncio.TimeoutError) as error:
|
||||
raise HTTPException(504, "Analysis request timed out waiting for budget") from error
|
||||
admitted.set()
|
||||
yield
|
||||
except TimeoutError as error:
|
||||
except (TimeoutError, asyncio.TimeoutError) as error:
|
||||
raise HTTPException(504, "Analysis request timed out waiting for budget or model output") from error
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ dependencies = [
|
|||
"click>=8.0.0,<9.0",
|
||||
"jinja2>=3.1.6,<4.0",
|
||||
"aiohttp>=3.14.2,<4.0",
|
||||
"async-timeout>=4.0.3,<6.0; python_version < '3.11'",
|
||||
"pydantic>=2.11.0,<3.0.0; python_version < '3.14'",
|
||||
"pydantic>=2.12.0,<3.0.0; python_version >= '3.14'",
|
||||
"pydantic-settings>=2.14.1,<3.0",
|
||||
|
|
|
|||
|
|
@ -621,6 +621,101 @@ async def test_renewal_preserves_the_original_failure_while_request_cleanup_is_p
|
|||
assert db.stored.reservations == (hold,)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("stalled", ("budget", "model", "budget_wait"))
|
||||
async def test_request_deadline_expiry_returns_gateway_timeout(stalled: str, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
import asyncio
|
||||
|
||||
from litellm.proxy.lens import inference
|
||||
from litellm.proxy.lens.inference import BUDGET_LEASE, reserve_amount, reserved_budget
|
||||
from litellm.proxy.lens.models import BudgetReservation
|
||||
from litellm.proxy.lens.repository import LensRepository
|
||||
from tests.unit.proxy.lens.test_endpoints import ResultDatabase
|
||||
from tests.unit.proxy.lens.test_state import NOW, lens
|
||||
|
||||
budget_deadline: Final = inference.BUDGET_WAIT_TIMEOUT
|
||||
request_deadline: Final = budget_deadline * 2 if stalled == "budget_wait" else budget_deadline / 2
|
||||
monkeypatch.setattr(litellm, "request_timeout", request_deadline)
|
||||
loop: Final = asyncio.get_running_loop()
|
||||
monkeypatch.setattr(loop, "time", lambda: 0.0)
|
||||
loop.call_soon(monkeypatch.setattr, loop, "time", lambda: min(request_deadline, budget_deadline) + 1)
|
||||
db: Final = ResultDatabase(lens())
|
||||
hold: Final = BudgetReservation(
|
||||
id="active", job_id="run", amount=90, month=db.stored.budget_month, expires_at=NOW + BUDGET_LEASE
|
||||
)
|
||||
admitted: Final = asyncio.Event()
|
||||
|
||||
with pytest.raises(HTTPException) as error:
|
||||
async with reserved_budget(
|
||||
LensRepository(db),
|
||||
"lens",
|
||||
hold.id,
|
||||
(lambda e: reserve_amount(e, hold, NOW)) if stalled == "model" else (lambda e: e),
|
||||
admitted,
|
||||
):
|
||||
await asyncio.Event().wait()
|
||||
assert error.value.status_code == 504
|
||||
assert error.value.detail == (
|
||||
"Analysis request timed out waiting for budget"
|
||||
if stalled == "budget_wait"
|
||||
else "Analysis request timed out waiting for budget or model output"
|
||||
)
|
||||
assert isinstance(error.value.__cause__, asyncio.TimeoutError)
|
||||
assert admitted.is_set() is (stalled == "model")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_renewal_deadline_cancels_stalled_database_and_model(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
import asyncio
|
||||
from collections.abc import AsyncGenerator
|
||||
from contextlib import asynccontextmanager
|
||||
|
||||
from litellm.proxy.lens import inference
|
||||
from litellm.proxy.lens.repository import Database, LensRepository, Row
|
||||
|
||||
interval: Final = inference.BUDGET_RENEW_INTERVAL
|
||||
loop: Final = asyncio.get_running_loop()
|
||||
monkeypatch.setattr(loop, "time", lambda: 0.0)
|
||||
admitted: Final = asyncio.Event()
|
||||
database_cancelled: Final = asyncio.Event()
|
||||
model_cancelled: Final = asyncio.Event()
|
||||
|
||||
class StalledDatabase:
|
||||
async def query_raw(self, query: str, *args: object) -> tuple[Row, ...]:
|
||||
loop.call_soon(monkeypatch.setattr, loop, "time", lambda: 2 * (interval + 1))
|
||||
try:
|
||||
await asyncio.Event().wait()
|
||||
finally:
|
||||
database_cancelled.set()
|
||||
return ()
|
||||
|
||||
async def execute_raw(self, query: str, *args: object) -> int:
|
||||
pytest.fail("A stalled read cannot write")
|
||||
|
||||
@asynccontextmanager
|
||||
async def transaction(self) -> AsyncGenerator[Database]:
|
||||
yield self
|
||||
|
||||
async def model() -> tuple[ModelResponse, float | None]:
|
||||
admitted.set()
|
||||
loop.call_soon(monkeypatch.setattr, loop, "time", lambda: interval + 1)
|
||||
try:
|
||||
await asyncio.Event().wait()
|
||||
finally:
|
||||
model_cancelled.set()
|
||||
pytest.fail("The stalled model must be cancelled when renewal times out")
|
||||
|
||||
with pytest.raises(HTTPException) as error:
|
||||
await inference.model_with_renewal(
|
||||
model(), inference.renew_budget_reservation(LensRepository(StalledDatabase()), "lens", "active", admitted)
|
||||
)
|
||||
assert error.value.status_code == 503
|
||||
assert error.value.detail == "Analysis budget reservation renewal timed out"
|
||||
assert isinstance(error.value.__cause__, asyncio.TimeoutError)
|
||||
assert database_cancelled.is_set()
|
||||
assert model_cancelled.is_set()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_completed_paid_response_survives_simultaneous_renewal_failure() -> None:
|
||||
from litellm.proxy.lens.inference import model_with_renewal
|
||||
|
|
|
|||
2
uv.lock
generated
2
uv.lock
generated
|
|
@ -4504,6 +4504,7 @@ version = "1.105.0"
|
|||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "aiohttp" },
|
||||
{ name = "async-timeout", marker = "python_full_version < '3.11'" },
|
||||
{ name = "boto3" },
|
||||
{ name = "click" },
|
||||
{ name = "fastuuid" },
|
||||
|
|
@ -4757,6 +4758,7 @@ requires-dist = [
|
|||
{ name = "aiohttp", specifier = ">=3.14.2,<4.0" },
|
||||
{ name = "anthropic", extras = ["vertex"], marker = "extra == 'proxy-runtime'", specifier = ">=0.84.0,<1.0" },
|
||||
{ name = "apscheduler", marker = "extra == 'proxy'", specifier = ">=3.11.2,<4.0" },
|
||||
{ name = "async-timeout", marker = "python_full_version < '3.11'", specifier = ">=4.0.3,<6.0" },
|
||||
{ name = "audioread", marker = "extra == 'stt-nvidia-riva'", specifier = ">=3.0.1" },
|
||||
{ name = "aurelio-sdk", marker = "python_full_version < '3.14' and extra == 'semantic-router'", specifier = ">=0.0.19,<1.0" },
|
||||
{ name = "aws-sdk-bedrock-runtime", extras = ["awscrt"], marker = "python_full_version >= '3.12' and extra == 'bedrock-realtime'", specifier = ">=0.10.0,<0.12.0" },
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue