From eba51ec958089bb6bcaecd5121d1355da516e897 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Thu, 3 Sep 2026 11:35:30 -0700 Subject: [PATCH] test: point provider timeout tests at the fake endpoint server The fake OpenAI endpoint now also serves the Azure deployment, Anthropic /v1/messages and Bedrock converse URL shapes, sleeping for the slow-endpoint model, so the remaining timeout tests that raced a real provider with a 10 microsecond to 10 millisecond deadline get a deterministic timeout instead. test_hanging_request_azure drops its httpx.AsyncClient.send monkeypatch for the same slow deployment --- tests/_fake_openai_endpoint_server.py | 117 +++++++- tests/local_testing/test_embedding.py | 8 +- .../test_fake_openai_endpoint.py | 71 +++++ tests/local_testing/test_timeout.py | 270 +++++------------- 4 files changed, 257 insertions(+), 209 deletions(-) diff --git a/tests/_fake_openai_endpoint_server.py b/tests/_fake_openai_endpoint_server.py index d3af31aeef5..2e842ae2197 100644 --- a/tests/_fake_openai_endpoint_server.py +++ b/tests/_fake_openai_endpoint_server.py @@ -12,7 +12,11 @@ gets back a well-formed chat/text/embedding/moderation response with realistic ``usage`` so cost tracking and spend accounting still exercise their real code paths. The one behavioral special case mirrors the old hosted mock: a request whose ``model`` is ``429`` returns HTTP 429 so rate-limit and cooldown tests -still have something to trip on. +still have something to trip on, and a model named ``slow-endpoint`` sleeps past +any short client deadline so timeout tests get a deterministic timeout. The Azure +deployment, Anthropic ``/v1/messages`` and Bedrock ``converse`` URL shapes are +served too, so provider timeout tests can point ``api_base`` here instead of +racing a real provider. """ from __future__ import annotations @@ -21,7 +25,8 @@ import asyncio import json import time import uuid -from typing import AsyncIterator, Final +from collections.abc import AsyncIterator +from typing import Final import uvicorn from starlette.applications import Starlette @@ -65,6 +70,11 @@ def _requested_model(body: dict[str, object]) -> str: return model if isinstance(model, str) else "mock-model" +async def _sleep_if_slow(model: str) -> None: + if model == _SLOW_MODEL: + await asyncio.sleep(_SLOW_RESPONSE_SECONDS) + + def _wants_stream(body: dict[str, object]) -> bool: return body.get("stream") is True @@ -136,13 +146,10 @@ async def _chat_completion_stream(model: str, with_usage: bool) -> AsyncIterator yield "data: [DONE]\n\n" -async def chat_completions(request: Request) -> Response: - body = await _parse_body(request) - model = _requested_model(body) +async def _chat_completion_response(model: str, body: dict[str, object]) -> Response: if model == _RATE_LIMIT_MODEL: return _rate_limit_response(model) - if model == _SLOW_MODEL: - await asyncio.sleep(_SLOW_RESPONSE_SECONDS) + await _sleep_if_slow(model) if _wants_stream(body): return StreamingResponse( _chat_completion_stream(model, _wants_stream_usage(body)), @@ -151,6 +158,16 @@ async def chat_completions(request: Request) -> Response: return JSONResponse(_chat_completion_body(model)) +async def chat_completions(request: Request) -> Response: + body = await _parse_body(request) + return await _chat_completion_response(_requested_model(body), body) + + +async def azure_chat_completions(request: Request) -> Response: + body = await _parse_body(request) + return await _chat_completion_response(request.path_params["deployment"], body) + + def _text_completion_body(model: str) -> dict[str, object]: return { "id": f"cmpl-{uuid.uuid4().hex[:24]}", @@ -195,8 +212,7 @@ async def completions(request: Request) -> Response: model = _requested_model(body) if model == _RATE_LIMIT_MODEL: return _rate_limit_response(model) - if model == _SLOW_MODEL: - await asyncio.sleep(_SLOW_RESPONSE_SECONDS) + await _sleep_if_slow(model) if _wants_stream(body): return StreamingResponse( _text_completion_stream(model, _wants_stream_usage(body)), @@ -205,11 +221,8 @@ async def completions(request: Request) -> Response: return JSONResponse(_text_completion_body(model)) -async def embeddings(request: Request) -> Response: - body = await _parse_body(request) - model = _requested_model(body) - if model == _SLOW_MODEL: - await asyncio.sleep(_SLOW_RESPONSE_SECONDS) +async def _embeddings_response(model: str, body: dict[str, object]) -> Response: + await _sleep_if_slow(model) raw_input = body.get("input", "") count = len(raw_input) if isinstance(raw_input, list) else 1 return JSONResponse( @@ -222,6 +235,76 @@ async def embeddings(request: Request) -> Response: ) +async def embeddings(request: Request) -> Response: + body = await _parse_body(request) + return await _embeddings_response(_requested_model(body), body) + + +async def azure_embeddings(request: Request) -> Response: + body = await _parse_body(request) + return await _embeddings_response(request.path_params["deployment"], body) + + +def _anthropic_message_body(model: str) -> dict[str, object]: + return { + "id": f"msg_{uuid.uuid4().hex[:24]}", + "type": "message", + "role": "assistant", + "model": model, + "content": [{"type": "text", "text": _CANNED_CONTENT}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": _PROMPT_TOKENS, "output_tokens": _COMPLETION_TOKENS}, + } + + +async def _anthropic_message_stream(model: str) -> AsyncIterator[str]: + def event(name: str, payload: dict[str, object]) -> str: + return f"event: {name}\ndata: {json.dumps({'type': name} | payload)}\n\n" + + opening = _anthropic_message_body(model) | { + "content": [], + "stop_reason": None, + "usage": {"input_tokens": _PROMPT_TOKENS, "output_tokens": 0}, + } + yield event("message_start", {"message": opening}) + yield event("content_block_start", {"index": 0, "content_block": {"type": "text", "text": ""}}) + yield event("content_block_delta", {"index": 0, "delta": {"type": "text_delta", "text": _CANNED_CONTENT}}) + yield event("content_block_stop", {"index": 0}) + yield event( + "message_delta", + {"delta": {"stop_reason": "end_turn", "stop_sequence": None}, "usage": {"output_tokens": _COMPLETION_TOKENS}}, + ) + yield event("message_stop", {}) + + +async def anthropic_messages(request: Request) -> Response: + body = await _parse_body(request) + model = _requested_model(body) + await _sleep_if_slow(model) + if _wants_stream(body): + return StreamingResponse(_anthropic_message_stream(model), media_type="text/event-stream") + return JSONResponse(_anthropic_message_body(model)) + + +def _bedrock_converse_body() -> dict[str, object]: + return { + "output": {"message": {"role": "assistant", "content": [{"text": _CANNED_CONTENT}]}}, + "stopReason": "end_turn", + "usage": { + "inputTokens": _PROMPT_TOKENS, + "outputTokens": _COMPLETION_TOKENS, + "totalTokens": _PROMPT_TOKENS + _COMPLETION_TOKENS, + }, + "metrics": {"latencyMs": 1}, + } + + +async def bedrock_converse(request: Request) -> Response: + await _sleep_if_slow(request.path_params["model_id"]) + return JSONResponse(_bedrock_converse_body()) + + async def triton_embeddings(_request: Request) -> Response: return JSONResponse( { @@ -286,6 +369,12 @@ app = Starlette( Route("/v1/completions", completions, methods=["POST"]), Route("/embeddings", embeddings, methods=["POST"]), Route("/v1/embeddings", embeddings, methods=["POST"]), + Route("/openai/deployments/{deployment}/chat/completions", azure_chat_completions, methods=["POST"]), + Route("/openai/deployments/{deployment}/embeddings", azure_embeddings, methods=["POST"]), + Route("/openai/v1/chat/completions", chat_completions, methods=["POST"]), + Route("/openai/v1/embeddings", embeddings, methods=["POST"]), + Route("/v1/messages", anthropic_messages, methods=["POST"]), + Route("/model/{model_id}/converse", bedrock_converse, methods=["POST"]), Route("/triton/embeddings", triton_embeddings, methods=["POST"]), Route("/moderations", moderations, methods=["POST"]), Route("/v1/moderations", moderations, methods=["POST"]), diff --git a/tests/local_testing/test_embedding.py b/tests/local_testing/test_embedding.py index c119334da6f..140b90c0629 100644 --- a/tests/local_testing/test_embedding.py +++ b/tests/local_testing/test_embedding.py @@ -250,11 +250,15 @@ async def test_azure_ai_embedding_image(sync_mode): def test_openai_azure_embedding_timeouts(): try: response = embedding( - model="azure/text-embedding-ada-002", + model="azure/slow-endpoint", input=["good morning from litellm"], - timeout=0.00001, + api_base=FAKE_OPENAI_API_BASE, + api_key="fake-key", + api_version="2024-10-21", + timeout=0.5, ) print(response) + pytest.fail("Expected timeout error, the request returned instead") except openai.APITimeoutError: print("Good job got timeout error!") pass diff --git a/tests/local_testing/test_fake_openai_endpoint.py b/tests/local_testing/test_fake_openai_endpoint.py index d5236d3de1b..c28a8701646 100644 --- a/tests/local_testing/test_fake_openai_endpoint.py +++ b/tests/local_testing/test_fake_openai_endpoint.py @@ -94,6 +94,77 @@ def test_slow_model_blocks_past_client_timeout(): ) +_CHAT_BODY: Final = {"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5} +_SLOW_PROVIDER_ROUTES: Final = ( + ("/openai/deployments/slow-endpoint/chat/completions", _CHAT_BODY), + ("/openai/deployments/slow-endpoint/embeddings", {"input": "hi"}), + ("/v1/messages", _CHAT_BODY | {"model": "slow-endpoint"}), + ("/model/slow-endpoint/converse", _CHAT_BODY), +) + + +@pytest.mark.parametrize("path, body", _SLOW_PROVIDER_ROUTES) +def test_slow_model_blocks_past_client_timeout_on_provider_routes(path, body): + base: Final = ensure_fake_openai_endpoint() + with pytest.raises(httpx.TimeoutException): + httpx.post(f"{base}{path}", json=body, timeout=0.5) + + +def test_azure_deployment_route_answers_as_the_path_deployment(): + base: Final = ensure_fake_openai_endpoint() + response: Final = httpx.post( + f"{base}/openai/deployments/my-deployment/chat/completions", json=_CHAT_BODY, timeout=10 + ) + assert response.status_code == 200 + assert response.json()["model"] == "my-deployment" + assert response.json()["choices"][0]["message"]["content"] + + +def test_azure_deployment_embeddings_route_shape(): + base: Final = ensure_fake_openai_endpoint() + response: Final = httpx.post( + f"{base}/openai/deployments/my-embedding/embeddings", json={"input": ["a", "b"]}, timeout=10 + ) + assert response.status_code == 200 + assert response.json()["model"] == "my-embedding" + assert len(response.json()["data"]) == 2 + + +def test_anthropic_messages_route_shape(): + base: Final = ensure_fake_openai_endpoint() + response: Final = httpx.post(f"{base}/v1/messages", json=_CHAT_BODY | {"model": "claude-x"}, timeout=10) + assert response.status_code == 200 + body: Final = response.json() + assert body["type"] == "message" + assert body["model"] == "claude-x" + assert body["content"][0]["text"] + assert body["usage"] == {"input_tokens": 20, "output_tokens": 20} + + +def test_anthropic_messages_route_streams_a_complete_message(): + base: Final = ensure_fake_openai_endpoint() + response: Final = httpx.post( + f"{base}/v1/messages", json=_CHAT_BODY | {"model": "claude-x", "stream": True}, timeout=10 + ) + assert response.status_code == 200 + events: Final = re.findall(r"^event: (\S+)$", response.text, flags=re.MULTILINE) + assert events[0] == "message_start" + assert events[-1] == "message_stop" + assert "content_block_delta" in events + + +def test_bedrock_converse_route_shape(): + base: Final = ensure_fake_openai_endpoint() + response: Final = httpx.post( + f"{base}/model/anthropic.claude-haiku-4-5-20251001-v1:0/converse", json=_CHAT_BODY, timeout=10 + ) + assert response.status_code == 200 + body: Final = response.json() + assert body["output"]["message"]["content"][0]["text"] + assert body["stopReason"] == "end_turn" + assert body["usage"]["totalTokens"] == 40 + + def test_remote_env_base_resolves_to_local(monkeypatch): monkeypatch.setenv( "FAKE_OPENAI_API_BASE", diff --git a/tests/local_testing/test_timeout.py b/tests/local_testing/test_timeout.py index 784e2c73cd7..5aaf06be0ba 100644 --- a/tests/local_testing/test_timeout.py +++ b/tests/local_testing/test_timeout.py @@ -1,11 +1,7 @@ #### What this tests #### # This tests the timeout decorator -import os -import traceback - -import time -from litellm._uuid import uuid +from typing import Final import httpx import openai @@ -14,6 +10,32 @@ import pytest import litellm from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE +_ESSAY_MESSAGES: Final = [{"role": "user", "content": "hello, write a 20 pg essay"}] +_TIMEOUT_SECONDS: Final = 0.5 + + +def _slow_openai_deployment(model_name: str) -> dict: + return { + "model_name": model_name, + "litellm_params": { + "model": "openai/slow-endpoint", + "api_base": FAKE_OPENAI_API_BASE, + "api_key": "fake-key", + }, + } + + +def _slow_azure_deployment(model_name: str) -> dict: + return { + "model_name": model_name, + "litellm_params": { + "model": "azure/slow-endpoint", + "api_base": FAKE_OPENAI_API_BASE, + "api_key": "fake-key", + "api_version": "2024-10-21", + }, + } + @pytest.mark.parametrize( "model, provider", @@ -45,201 +67,73 @@ async def test_httpx_timeout(model, provider, sync_mode): def test_timeout(): - # this Will Raise a timeout litellm.set_verbose = False - try: - response = litellm.completion( - model="gpt-3.5-turbo", - timeout=0.01, - messages=[{"role": "user", "content": "hello, write a 20 pg essay"}], + with pytest.raises(openai.APITimeoutError): + litellm.completion( + model="openai/slow-endpoint", + messages=_ESSAY_MESSAGES, + api_base=FAKE_OPENAI_API_BASE, + api_key="fake-key", + timeout=_TIMEOUT_SECONDS, ) - except openai.APITimeoutError as e: - print( - "Passed: Raised correct exception. Got openai.APITimeoutError\nGood Job", e - ) - print(type(e)) - pass - except Exception as e: - pytest.fail( - f"Did not raise error `openai.APITimeoutError`. Instead raised error type: {type(e)}, Error: {e}" - ) - - -# test_timeout() def test_bedrock_timeout(): - # this Will Raise a timeout litellm.set_verbose = True - try: - response = litellm.completion( - model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", - timeout=0.01, - messages=[{"role": "user", "content": "hello, write a 20 pg essay"}], - ) - pytest.fail("Did not raise error `openai.APITimeoutError`") - except openai.APITimeoutError as e: - print( - "Passed: Raised correct exception. Got openai.APITimeoutError\nGood Job", e - ) - print(type(e)) - pass - except Exception as e: - pytest.fail( - f"Did not raise error `openai.APITimeoutError`. Instead raised error type: {type(e)}, Error: {e}" + with pytest.raises(openai.APITimeoutError): + litellm.completion( + model="bedrock/converse/slow-endpoint", + messages=_ESSAY_MESSAGES, + api_base=FAKE_OPENAI_API_BASE, + aws_access_key_id="fake-access-key", + aws_secret_access_key="fake-secret-key", + aws_region_name="us-east-1", + timeout=_TIMEOUT_SECONDS, ) -def test_hanging_request_azure(): - """ - Test that a slow Azure request properly raises APITimeoutError via the Router. - - Uses a mock to simulate a slow HTTP response so the timeout fires reliably, - rather than racing against real network latency. - """ +@pytest.mark.asyncio +async def test_hanging_request_azure(): litellm.set_verbose = True - import asyncio - from unittest.mock import AsyncMock, patch - - try: - router = litellm.Router( - model_list=[ - { - "model_name": "azure-gpt", - "litellm_params": { - "model": "azure/gpt-4.1-mini", - "api_base": os.environ["AZURE_AI_API_BASE"], - "api_key": os.environ["AZURE_AI_API_KEY"], - }, - }, - { - "model_name": "openai-gpt", - "litellm_params": {"model": "gpt-3.5-turbo"}, - }, - ], - num_retries=0, + router = litellm.Router( + model_list=[_slow_azure_deployment("azure-gpt"), _slow_openai_deployment("openai-gpt")], + num_retries=0, + ) + with pytest.raises(openai.APITimeoutError): + await router.acompletion( + model="azure-gpt", + messages=[{"role": "user", "content": "what color is red"}], + timeout=_TIMEOUT_SECONDS, ) - encoded = litellm.utils.encode(model="gpt-3.5-turbo", text="blue")[0] - - original_send = httpx.AsyncClient.send - - async def _slow_send(self, request, *args, **kwargs): - await asyncio.sleep(5) - return await original_send(self, request, *args, **kwargs) - - async def _test(): - with patch.object(httpx.AsyncClient, "send", new=_slow_send): - response = await router.acompletion( - model="azure-gpt", - messages=[ - { - "role": "user", - "content": f"what color is red {uuid.uuid4()}", - } - ], - logit_bias={encoded: 100}, - timeout=0.01, - ) - print(response) - return response - - response = asyncio.run(_test()) - - if response.choices[0].message.content is not None: - pytest.fail("Got a response, expected a timeout") - except openai.APITimeoutError as e: - print( - "Passed: Raised correct exception. Got openai.APITimeoutError\nGood Job", e - ) - print(type(e)) - pass - except Exception as e: - pytest.fail( - f"Did not raise error `openai.APITimeoutError`. Instead raised error type: {type(e)}, Error: {e}" - ) - - -# test_hanging_request_azure() - def test_hanging_request_openai(): litellm.set_verbose = True - try: - router = litellm.Router( - model_list=[ - { - "model_name": "azure-gpt", - "litellm_params": { - "model": "azure/gpt-4.1-mini", - "api_base": os.environ["AZURE_AI_API_BASE"], - "api_key": os.environ["AZURE_AI_API_KEY"], - }, - }, - { - "model_name": "openai-gpt", - "litellm_params": {"model": "gpt-3.5-turbo"}, - }, - ], - num_retries=0, - ) - - encoded = litellm.utils.encode(model="gpt-3.5-turbo", text="blue")[0] - response = router.completion( + router = litellm.Router( + model_list=[_slow_azure_deployment("azure-gpt"), _slow_openai_deployment("openai-gpt")], + num_retries=0, + ) + with pytest.raises(openai.APITimeoutError): + router.completion( model="openai-gpt", messages=[{"role": "user", "content": "what color is red"}], - logit_bias={encoded: 100}, - timeout=0.01, + timeout=_TIMEOUT_SECONDS, ) - print(response) - - if response.choices[0].message.content is not None: - pytest.fail("Got a response, expected a timeout") - except openai.APITimeoutError as e: - print( - "Passed: Raised correct exception. Got openai.APITimeoutError\nGood Job", e - ) - print(type(e)) - pass - except Exception as e: - pytest.fail( - f"Did not raise error `openai.APITimeoutError`. Instead raised error type: {type(e)}, Error: {e}" - ) - - -# test_hanging_request_openai() - -# test_timeout() def test_timeout_streaming(): - # this Will Raise a timeout litellm.set_verbose = False - try: + with pytest.raises(openai.APITimeoutError): response = litellm.completion( model="openai/slow-endpoint", - messages=[{"role": "user", "content": "hello, write a 20 pg essay"}], + messages=_ESSAY_MESSAGES, api_base=FAKE_OPENAI_API_BASE, api_key="fake-key", - timeout=0.5, + timeout=_TIMEOUT_SECONDS, stream=True, ) for chunk in response: print(chunk) - pytest.fail("Did not raise error `openai.APITimeoutError`. The stream completed instead") - except openai.APITimeoutError as e: - print( - "Passed: Raised correct exception. Got openai.APITimeoutError\nGood Job", e - ) - print(type(e)) - pass - except Exception as e: - pytest.fail( - f"Did not raise error `openai.APITimeoutError`. Instead raised error type: {type(e)}, Error: {e}" - ) - - -# test_timeout_streaming() @pytest.mark.skip(reason="local test") @@ -273,32 +167,22 @@ def test_timeout_ollama(): @pytest.mark.asyncio async def test_anthropic_timeout(streaming, sync_mode): litellm.set_verbose = False - - try: + request: Final = { + "model": "anthropic/slow-endpoint", + "messages": _ESSAY_MESSAGES, + "api_base": FAKE_OPENAI_API_BASE, + "api_key": "fake-key", + "timeout": _TIMEOUT_SECONDS, + "stream": streaming, + } + with pytest.raises(openai.APITimeoutError): if sync_mode: - response = litellm.completion( - model="claude-sonnet-4-5-20250929", - timeout=0.01, - messages=[{"role": "user", "content": "hello, write a 20 pg essay"}], - stream=streaming, - ) + response = litellm.completion(**request) if isinstance(response, litellm.CustomStreamWrapper): - for chunk in response: + for _ in response: pass else: - response = await litellm.acompletion( - model="claude-sonnet-4-5-20250929", - timeout=0.01, - messages=[{"role": "user", "content": "hello, write a 20 pg essay"}], - stream=streaming, - ) + response = await litellm.acompletion(**request) if isinstance(response, litellm.CustomStreamWrapper): - async for chunk in response: + async for _ in response: pass - pytest.fail("Did not raise error `openai.APITimeoutError`") - except openai.APITimeoutError as e: - print( - "Passed: Raised correct exception. Got openai.APITimeoutError\nGood Job", e - ) - print(type(e)) - pass