From f9b4a76065ae5dee9dfa38c036b7eb252d0b1398 Mon Sep 17 00:00:00 2001 From: tan Date: Wed, 18 Mar 2026 06:51:51 +0530 Subject: [PATCH] update tests, use mocked AsyncHTTPHandler.post --- .../test_anthropic_timeout.py | 398 ++++++++---------- 1 file changed, 173 insertions(+), 225 deletions(-) diff --git a/tests/test_litellm/proxy/pass_through_endpoints/test_anthropic_timeout.py b/tests/test_litellm/proxy/pass_through_endpoints/test_anthropic_timeout.py index 3e041d75c4c..7d8ccd6c5e7 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/test_anthropic_timeout.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/test_anthropic_timeout.py @@ -3,23 +3,20 @@ Tests that the timeout set via router_settings (which mirrors --request_timeout, litellm_settings.request_timeout, or router_settings.timeout) actually reaches the HTTP client in the Anthropic adapter-based messages path. -Uses a local aiohttp mock server that delays its response to verify that: +Uses mocked httpx calls to verify that: - A short timeout causes the request to fail with a timeout error. - A sufficiently long timeout allows the request to succeed. - Streaming requests honour the same timeout for the initial connection. - -Each test loads its configuration from a temporary YAML file, mirroring -how the proxy initialises the Router from config. +- stream_timeout takes precedence over timeout for streaming requests. """ -import asyncio import os import sys -import textwrap +from unittest.mock import AsyncMock, MagicMock, patch +import httpx import pytest import yaml -from aiohttp import web sys.path.insert( 0, os.path.abspath("../../../..") @@ -27,51 +24,40 @@ sys.path.insert( from litellm import Router +MOCK_TARGET = "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post" + # --------------------------------------------------------------------------- -# Mock Anthropic /v1/messages endpoint with a fixed delay +# Mock response helpers # --------------------------------------------------------------------------- -MOCK_DELAY_SECONDS = 3 + +def _mock_anthropic_response() -> MagicMock: + """Return a mock httpx.Response for a non-streaming Anthropic messages call.""" + response_body = { + "id": "msg_test_timeout_123", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [{"type": "text", "text": "Hello from mock server"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": { + "input_tokens": 10, + "output_tokens": 5, + }, + } + mock_resp = MagicMock(spec=httpx.Response) + mock_resp.status_code = 200 + mock_resp.json.return_value = response_body + mock_resp.text = str(response_body) + mock_resp.headers = {"content-type": "application/json"} + return mock_resp -async def _delayed_messages_handler(request: web.Request) -> web.Response: - """Return a minimal valid Anthropic messages response after a delay.""" - await asyncio.sleep(MOCK_DELAY_SECONDS) - return web.json_response( - { - "id": "msg_test_timeout_123", - "type": "message", - "role": "assistant", - "model": "claude-sonnet-4-5-20250929", - "content": [{"type": "text", "text": "Hello from mock server"}], - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": { - "input_tokens": 10, - "output_tokens": 5, - }, - } - ) - - -async def _delayed_streaming_messages_handler( - request: web.Request, -) -> web.StreamResponse: - """Return a streaming SSE Anthropic messages response after a delay. - - The delay happens *before* any bytes are sent, so the httpx timeout - (which covers the period until response headers arrive) should trigger. - """ - await asyncio.sleep(MOCK_DELAY_SECONDS) - - response = web.StreamResponse( - status=200, - headers={"Content-Type": "text/event-stream"}, - ) - await response.prepare(request) - - events = [ +def _mock_streaming_response() -> MagicMock: + """Return a mock httpx.Response for a streaming Anthropic messages call.""" + sse_events = [ 'event: message_start\ndata: {"type":"message_start","message":{"id":"msg_stream_timeout_123","type":"message","role":"assistant","model":"claude-sonnet-4-5-20250929","content":[],"stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":10,"output_tokens":0}}}\n\n', 'event: content_block_start\ndata: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}\n\n', 'event: content_block_delta\ndata: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"Hello from streaming mock"}}\n\n', @@ -79,39 +65,16 @@ async def _delayed_streaming_messages_handler( 'event: message_delta\ndata: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"output_tokens":5}}\n\n', 'event: message_stop\ndata: {"type":"message_stop"}\n\n', ] - for event in events: - await response.write(event.encode()) - await response.write_eof() - return response + async def _aiter_bytes(): + for event in sse_events: + yield event.encode() - -@pytest.fixture() -async def mock_anthropic_server(): - """Start a local HTTP server that mimics a slow Anthropic API.""" - app = web.Application() - app.router.add_post("/v1/messages", _delayed_messages_handler) - runner = web.AppRunner(app) - await runner.setup() - site = web.TCPSite(runner, "127.0.0.1", 0) - await site.start() - port = site._server.sockets[0].getsockname()[1] - yield f"http://127.0.0.1:{port}" - await runner.cleanup() - - -@pytest.fixture() -async def mock_anthropic_streaming_server(): - """Start a local HTTP server that mimics a slow *streaming* Anthropic API.""" - app = web.Application() - app.router.add_post("/v1/messages", _delayed_streaming_messages_handler) - runner = web.AppRunner(app) - await runner.setup() - site = web.TCPSite(runner, "127.0.0.1", 0) - await site.start() - port = site._server.sockets[0].getsockname()[1] - yield f"http://127.0.0.1:{port}" - await runner.cleanup() + mock_resp = MagicMock(spec=httpx.Response) + mock_resp.status_code = 200 + mock_resp.headers = {"content-type": "text/event-stream"} + mock_resp.aiter_bytes = _aiter_bytes + return mock_resp # --------------------------------------------------------------------------- @@ -130,75 +93,83 @@ def _router_from_yaml(yaml_text: str) -> Router: return Router(model_list=model_list, **router_settings) +def _make_config(**router_settings) -> str: + """Build a YAML config string with the given router_settings.""" + config = { + "model_list": [ + { + "model_name": "test-model", + "litellm_params": { + "model": "anthropic/claude-sonnet-4-5-20250929", + "api_key": "fake-key", + "api_base": "http://mock-api.local", + }, + } + ], + "router_settings": router_settings, + } + return yaml.dump(config) + + # --------------------------------------------------------------------------- -# Tests +# Non-streaming tests # --------------------------------------------------------------------------- @pytest.mark.asyncio -async def test_anthropic_messages_timeout_too_short(mock_anthropic_server): +async def test_anthropic_messages_timeout_too_short(): """ - When the router timeout (1 s) is shorter than the server delay (3 s), - the request must raise a timeout error. + When the router timeout (1 s) is configured and the HTTP call raises a + timeout exception, the error must propagate to the caller. """ - config_yaml = textwrap.dedent( - f"""\ - model_list: - - model_name: test-model - litellm_params: - model: anthropic/claude-sonnet-4-5-20250929 - api_key: fake-key - api_base: "{mock_anthropic_server}" - router_settings: - timeout: 1 - num_retries: 0 - """ - ) + config_yaml = _make_config(timeout=1, num_retries=0) router = _router_from_yaml(config_yaml) - with pytest.raises(Exception) as exc_info: - await router.aanthropic_messages( + with patch(MOCK_TARGET, new_callable=AsyncMock) as mock_post: + mock_post.side_effect = httpx.TimeoutException("Read timeout") + + with pytest.raises(Exception) as exc_info: + await router.aanthropic_messages( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + max_tokens=10, + ) + + exc = exc_info.value + assert ( + "timeout" in type(exc).__name__.lower() or "timeout" in str(exc).lower() + ), f"Expected a timeout-related exception, got {type(exc).__name__}: {exc}" + + mock_post.assert_called_once() + _, kwargs = mock_post.call_args + assert kwargs["timeout"] == 1 + + +@pytest.mark.asyncio +async def test_anthropic_messages_timeout_sufficient(): + """ + When the router timeout (10 s) is configured and the HTTP call succeeds, + the response must be returned correctly with the right timeout passed. + """ + config_yaml = _make_config(timeout=10, num_retries=0) + router = _router_from_yaml(config_yaml) + + with patch(MOCK_TARGET, new_callable=AsyncMock) as mock_post: + mock_post.return_value = _mock_anthropic_response() + + response = await router.aanthropic_messages( model="test-model", messages=[{"role": "user", "content": "hi"}], max_tokens=10, ) - exc = exc_info.value - assert ( - "timeout" in type(exc).__name__.lower() or "timeout" in str(exc).lower() - ), f"Expected a timeout-related exception, got {type(exc).__name__}: {exc}" - - -@pytest.mark.asyncio -async def test_anthropic_messages_timeout_sufficient(mock_anthropic_server): - """ - When the router timeout (10 s) is longer than the server delay (3 s), - the request must succeed and return a valid response. - """ - config_yaml = textwrap.dedent( - f"""\ - model_list: - - model_name: test-model - litellm_params: - model: anthropic/claude-sonnet-4-5-20250929 - api_key: fake-key - api_base: "{mock_anthropic_server}" - router_settings: - timeout: 10 - num_retries: 0 - """ - ) - router = _router_from_yaml(config_yaml) - - response = await router.aanthropic_messages( - model="test-model", - messages=[{"role": "user", "content": "hi"}], - max_tokens=10, - ) - assert response["id"] == "msg_test_timeout_123" assert response["content"][0]["text"] == "Hello from mock server" + mock_post.assert_called_once() + _, kwargs = mock_post.call_args + assert kwargs["timeout"] == 10 + # --------------------------------------------------------------------------- # Streaming tests @@ -206,121 +177,98 @@ async def test_anthropic_messages_timeout_sufficient(mock_anthropic_server): @pytest.mark.asyncio -async def test_anthropic_streaming_timeout_too_short(mock_anthropic_streaming_server): +async def test_anthropic_streaming_timeout_too_short(): """ - When the router timeout (1 s) is shorter than the server delay (3 s), - a *streaming* request must raise a timeout error (the timeout fires - before response headers arrive). + When the router timeout (1 s) is configured and the streaming HTTP call + raises a timeout exception, the error must propagate to the caller. """ - config_yaml = textwrap.dedent( - f"""\ - model_list: - - model_name: test-model - litellm_params: - model: anthropic/claude-sonnet-4-5-20250929 - api_key: fake-key - api_base: "{mock_anthropic_streaming_server}" - router_settings: - timeout: 1 - num_retries: 0 - """ - ) + config_yaml = _make_config(timeout=1, num_retries=0) router = _router_from_yaml(config_yaml) - with pytest.raises(Exception) as exc_info: - resp = await router.aanthropic_messages( + with patch(MOCK_TARGET, new_callable=AsyncMock) as mock_post: + mock_post.side_effect = httpx.TimeoutException("Read timeout") + + with pytest.raises(Exception) as exc_info: + resp = await router.aanthropic_messages( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + max_tokens=10, + stream=True, + ) + if hasattr(resp, "__aiter__"): + async for _ in resp: + break + + exc = exc_info.value + assert ( + "timeout" in type(exc).__name__.lower() or "timeout" in str(exc).lower() + ), f"Expected a timeout-related exception, got {type(exc).__name__}: {exc}" + + mock_post.assert_called_once() + _, kwargs = mock_post.call_args + assert kwargs["timeout"] == 1 + + +@pytest.mark.asyncio +async def test_anthropic_streaming_timeout_sufficient(): + """ + When the router timeout (10 s) is configured and the streaming HTTP call + succeeds, the response chunks must contain the expected text. + """ + config_yaml = _make_config(timeout=10, num_retries=0) + router = _router_from_yaml(config_yaml) + + with patch(MOCK_TARGET, new_callable=AsyncMock) as mock_post: + mock_post.return_value = _mock_streaming_response() + + response = await router.aanthropic_messages( model="test-model", messages=[{"role": "user", "content": "hi"}], max_tokens=10, stream=True, ) - # If the call itself doesn't raise (some implementations return an - # async iterator), consuming the first chunk should trigger it. - if hasattr(resp, "__aiter__"): - async for _ in resp: - break - exc = exc_info.value - assert ( - "timeout" in type(exc).__name__.lower() or "timeout" in str(exc).lower() - ), f"Expected a timeout-related exception, got {type(exc).__name__}: {exc}" + chunks = [] + async for chunk in response: + chunks.append(chunk) - -@pytest.mark.asyncio -async def test_anthropic_streaming_timeout_sufficient(mock_anthropic_streaming_server): - """ - When the router timeout (10 s) is longer than the server delay (3 s), - a streaming request must succeed and yield SSE chunks. - """ - config_yaml = textwrap.dedent( - f"""\ - model_list: - - model_name: test-model - litellm_params: - model: anthropic/claude-sonnet-4-5-20250929 - api_key: fake-key - api_base: "{mock_anthropic_streaming_server}" - router_settings: - timeout: 10 - num_retries: 0 - """ - ) - router = _router_from_yaml(config_yaml) - - response = await router.aanthropic_messages( - model="test-model", - messages=[{"role": "user", "content": "hi"}], - max_tokens=10, - stream=True, - ) - - # The response is an async generator of SSE chunks; drain it. - chunks = [] - async for chunk in response: - chunks.append(chunk) - - # We should have received at least one chunk containing our test text. joined = b"".join(c if isinstance(c, bytes) else c.encode() for c in chunks) assert b"Hello from streaming mock" in joined + mock_post.assert_called_once() + _, kwargs = mock_post.call_args + assert kwargs["timeout"] == 10 + @pytest.mark.asyncio -async def test_anthropic_streaming_stream_timeout_too_short( - mock_anthropic_streaming_server, -): +async def test_anthropic_streaming_stream_timeout_too_short(): """ When stream_timeout (1 s) is set explicitly and is shorter than the - server delay (3 s), the streaming request must time out — even if the - general timeout is generous. + general timeout (30 s), stream_timeout must take precedence for streaming + requests. """ - config_yaml = textwrap.dedent( - f"""\ - model_list: - - model_name: test-model - litellm_params: - model: anthropic/claude-sonnet-4-5-20250929 - api_key: fake-key - api_base: "{mock_anthropic_streaming_server}" - router_settings: - timeout: 30 - stream_timeout: 1 - num_retries: 0 - """ - ) + config_yaml = _make_config(timeout=30, stream_timeout=1, num_retries=0) router = _router_from_yaml(config_yaml) - with pytest.raises(Exception) as exc_info: - resp = await router.aanthropic_messages( - model="test-model", - messages=[{"role": "user", "content": "hi"}], - max_tokens=10, - stream=True, - ) - if hasattr(resp, "__aiter__"): - async for _ in resp: - break + with patch(MOCK_TARGET, new_callable=AsyncMock) as mock_post: + mock_post.side_effect = httpx.TimeoutException("Read timeout") - exc = exc_info.value - assert ( - "timeout" in type(exc).__name__.lower() or "timeout" in str(exc).lower() - ), f"Expected a timeout-related exception, got {type(exc).__name__}: {exc}" + with pytest.raises(Exception) as exc_info: + resp = await router.aanthropic_messages( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + max_tokens=10, + stream=True, + ) + if hasattr(resp, "__aiter__"): + async for _ in resp: + break + + exc = exc_info.value + assert ( + "timeout" in type(exc).__name__.lower() or "timeout" in str(exc).lower() + ), f"Expected a timeout-related exception, got {type(exc).__name__}: {exc}" + + mock_post.assert_called_once() + _, kwargs = mock_post.call_args + assert kwargs["timeout"] == 1