diff --git a/tests/e2e/coverage_registry/llm_conversational.yaml b/tests/e2e/coverage_registry/llm_conversational.yaml index be8a291c6fc..40d766e7ce5 100644 --- a/tests/e2e/coverage_registry/llm_conversational.yaml +++ b/tests/e2e/coverage_registry/llm_conversational.yaml @@ -29,6 +29,7 @@ - {id: llm.chat_completions.vertex.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Gemini vision"} - {id: llm.chat_completions.vertex.prompt_cache_5m.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: vertex, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Vertex Gemini prompt caching"} - {id: llm.chat_completions.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Azure OpenAI deployments"} +- {id: llm.chat_completions.azure_openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Azure OpenAI"} - {id: llm.chat_completions.azure_openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Azure OpenAI function_calling"} - {id: llm.chat_completions.azure_foundry.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: azure_foundry, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Azure Foundry (azure_ai); newer, smoke"} - {id: llm.messages.anthropic.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Core endpoint; Anthropic Messages native"} diff --git a/tests/e2e/docker-compose.yml b/tests/e2e/docker-compose.yml index e2bb6ca8933..4e036b20462 100644 --- a/tests/e2e/docker-compose.yml +++ b/tests/e2e/docker-compose.yml @@ -66,6 +66,12 @@ configs: model: openai/text-embedding-3-small api_key: os.environ/OPENAI_API_KEY + - model_name: azure-gpt-5.4-mini + litellm_params: + model: azure/gpt-5.4-mini + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY + services: litellm: image: ghcr.io/berriai/litellm:main-latest diff --git a/tests/e2e/e2e_http.py b/tests/e2e/e2e_http.py index ff296969079..53d3c52d61a 100644 --- a/tests/e2e/e2e_http.py +++ b/tests/e2e/e2e_http.py @@ -121,6 +121,7 @@ class StreamingResponse(BaseModel): headers: dict[str, str] = {} body: str chunks: int = 0 # streamed events (0 for non-streaming) + events: tuple[str, ...] = () # SSE "data:" payloads, prefix stripped (streaming only) # First in-stream error event, if any. A streamed call commits its HTTP 200 # before the upstream completes, so upstream failures (e.g. insufficient # quota) arrive as SSE error events inside an otherwise-successful response; @@ -293,20 +294,22 @@ def _streaming_outcome(resp: requests.Response, stream: bool) -> StreamingRespon headers=headers, body=resp.text, ) - lines = cast("Iterator[bytes]", resp.iter_lines()) - chunks = 0 - stream_error: str | None = None - for line in lines: - if not line: - continue - chunks += 1 - if stream_error is None and ( - line.startswith(b"event: error") - or b'"type":"error"' in line - or b'"type": "error"' in line - or line.startswith(b'data: {"error"') - ): - stream_error = line.decode(errors="replace")[:300] + raw_lines = tuple( + line.decode(errors="replace") + for line in cast("Iterator[bytes]", resp.iter_lines()) + if line + ) + stream_error = next( + ( + line[:300] + for line in raw_lines + if line.startswith("event: error") + or '"type":"error"' in line + or '"type": "error"' in line + or line.startswith('data: {"error"') + ), + None, + ) return StreamingResponse( status_code=resp.status_code, call_id=call_id, @@ -314,7 +317,12 @@ def _streaming_outcome(resp: requests.Response, stream: bool) -> StreamingRespon content_type=content_type, headers=headers, body="", - chunks=chunks, + chunks=len(raw_lines), + events=tuple( + line.removeprefix("data:").strip() + for line in raw_lines + if line.startswith("data:") + ), stream_error=stream_error, ) diff --git a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py index 5cc4ff308fa..efcd6fddc2f 100644 --- a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py +++ b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py @@ -4,9 +4,11 @@ GH #28991 broke /chat/completions (and /responses) for most models on some releases: a clean 200 came back but with no real completion. A status check alone would not have caught it, so each case here asserts the product promise - a non-empty assistant message and a real model name in the body - across the -three providers wired into the gateway config (OpenAI, Anthropic, Gemini). A -regression that empties the completion for any provider fails that provider's -row here. +providers wired into the gateway config (OpenAI, Anthropic, Gemini, Azure +OpenAI). A regression that empties the completion for any provider fails that +provider's row here. The Azure OpenAI streaming case applies the same standard +to the SSE path: every data event must parse as a chat.completion.chunk and the +deltas must reassemble into real text, not just count as a 200 with chunks. """ from __future__ import annotations @@ -15,15 +17,18 @@ import pytest from e2e_config import unique_marker from e2e_http import unwrap -from models import ChatBody, ChatMessage +from models import ChatBody, ChatMessage, ChatStreamChunk from passthrough_client import PassthroughClient pytestmark = pytest.mark.e2e +AZURE_CHAT_MODEL = "azure-gpt-5.4-mini" + CHAT_MODELS: tuple[tuple[str, str], ...] = ( ("gpt-5.5", "openai"), ("claude-haiku-4-5", "anthropic"), ("gemini-2.5-flash", "gemini"), + (AZURE_CHAT_MODEL, "azure_openai"), ) @@ -37,6 +42,7 @@ class TestChatCompletionsRegression: "llm.chat_completions.openai.basic.nonstream.works", "llm.chat_completions.anthropic.basic.nonstream.works", "llm.chat_completions.vertex.basic.nonstream.works", + "llm.chat_completions.azure_openai.basic.nonstream.works", exercised_on=[], ) def test_chat_returns_real_completion( @@ -68,3 +74,62 @@ class TestChatCompletionsRegression: assert ( message is not None and message.content and message.content.strip() ), f"{model} ({route}): 200 with an empty completion (#28991): {response}" + + @pytest.mark.covers("llm.chat_completions.azure_openai.basic.stream.works") + def test_azure_openai_stream_returns_real_completion( + self, client: PassthroughClient, scoped_key: str + ) -> None: + result = client.gateway.chat_stream( + scoped_key, + ChatBody( + model=AZURE_CHAT_MODEL, + messages=[ + ChatMessage( + role="user", + content=f"reply with one word {unique_marker()}", + ) + ], + max_tokens=512, + stream=True, + ), + ) + + assert result.ok, ( + f"{AZURE_CHAT_MODEL}: stream failed with status " + f"{result.status_code}: {result.body[:300]}" + ) + assert result.is_streaming, ( + f"{AZURE_CHAT_MODEL}: expected text/event-stream, got " + f"{result.content_type}: {result.body[:300]}" + ) + assert result.stream_error is None, ( + f"{AZURE_CHAT_MODEL}: 200 stream carried an error event: " + f"{result.stream_error}" + ) + assert result.events, f"{AZURE_CHAT_MODEL}: stream carried no SSE data events" + assert result.events[-1] == "[DONE]", ( + f"{AZURE_CHAT_MODEL}: stream did not terminate with [DONE]: " + f"{result.events[-1][:200]}" + ) + + chunks = [ + ChatStreamChunk.model_validate_json(event) for event in result.events[:-1] + ] + assert chunks, f"{AZURE_CHAT_MODEL}: stream held only the [DONE] sentinel" + assert all( + chunk.object == "chat.completion.chunk" for chunk in chunks + ), f"{AZURE_CHAT_MODEL}: malformed chunk object types: {result.events[:5]}" + assert any( + chunk.model for chunk in chunks + ), f"{AZURE_CHAT_MODEL}: no chunk carried a model name: {result.events[:5]}" + + content = "".join( + choice.delta.content or "" + for chunk in chunks + for choice in chunk.choices + if choice.delta is not None + ) + assert content.strip(), ( + f"{AZURE_CHAT_MODEL}: stream chunks reassembled to an empty " + f"completion (#28991): {result.events[:5]}" + ) diff --git a/tests/e2e/models.py b/tests/e2e/models.py index 4140967f3e0..6e3bd8ad554 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -191,6 +191,21 @@ class ChatResponse(BaseModel): service_tier: str | None = None +class ChatStreamDelta(BaseModel): + content: str | None = None + + +class ChatStreamChoice(BaseModel): + delta: ChatStreamDelta | None = None + + +class ChatStreamChunk(BaseModel): + id: str | None = None + object: str | None = None + model: str | None = None + choices: list[ChatStreamChoice] = [] + + class EmbedBody(BaseModel): model: str input: str