test(e2e): cover /chat/completions on Azure OpenAI, streaming and non-streaming

This commit is contained in:
mateo-berri 2026-07-15 16:11:45 -07:00
parent c6d49a85b2
commit e7c3f7bc75
5 changed files with 114 additions and 19 deletions

View file

@ -29,6 +29,7 @@
- {id: llm.chat_completions.vertex.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Gemini vision"}
- {id: llm.chat_completions.vertex.prompt_cache_5m.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: vertex, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Vertex Gemini prompt caching"}
- {id: llm.chat_completions.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Azure OpenAI deployments"}
- {id: llm.chat_completions.azure_openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Azure OpenAI"}
- {id: llm.chat_completions.azure_openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Azure OpenAI function_calling"}
- {id: llm.chat_completions.azure_foundry.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: azure_foundry, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Azure Foundry (azure_ai); newer, smoke"}
- {id: llm.messages.anthropic.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Core endpoint; Anthropic Messages native"}

View file

@ -66,6 +66,12 @@ configs:
model: openai/text-embedding-3-small
api_key: os.environ/OPENAI_API_KEY
- model_name: azure-gpt-5.4-mini
litellm_params:
model: azure/gpt-5.4-mini
api_base: os.environ/AZURE_API_BASE
api_key: os.environ/AZURE_API_KEY
services:
litellm:
image: ghcr.io/berriai/litellm:main-latest

View file

@ -121,6 +121,7 @@ class StreamingResponse(BaseModel):
headers: dict[str, str] = {}
body: str
chunks: int = 0 # streamed events (0 for non-streaming)
events: tuple[str, ...] = () # SSE "data:" payloads, prefix stripped (streaming only)
# First in-stream error event, if any. A streamed call commits its HTTP 200
# before the upstream completes, so upstream failures (e.g. insufficient
# quota) arrive as SSE error events inside an otherwise-successful response;
@ -293,20 +294,22 @@ def _streaming_outcome(resp: requests.Response, stream: bool) -> StreamingRespon
headers=headers,
body=resp.text,
)
lines = cast("Iterator[bytes]", resp.iter_lines())
chunks = 0
stream_error: str | None = None
for line in lines:
if not line:
continue
chunks += 1
if stream_error is None and (
line.startswith(b"event: error")
or b'"type":"error"' in line
or b'"type": "error"' in line
or line.startswith(b'data: {"error"')
):
stream_error = line.decode(errors="replace")[:300]
raw_lines = tuple(
line.decode(errors="replace")
for line in cast("Iterator[bytes]", resp.iter_lines())
if line
)
stream_error = next(
(
line[:300]
for line in raw_lines
if line.startswith("event: error")
or '"type":"error"' in line
or '"type": "error"' in line
or line.startswith('data: {"error"')
),
None,
)
return StreamingResponse(
status_code=resp.status_code,
call_id=call_id,
@ -314,7 +317,12 @@ def _streaming_outcome(resp: requests.Response, stream: bool) -> StreamingRespon
content_type=content_type,
headers=headers,
body="<streamed>",
chunks=chunks,
chunks=len(raw_lines),
events=tuple(
line.removeprefix("data:").strip()
for line in raw_lines
if line.startswith("data:")
),
stream_error=stream_error,
)

View file

@ -4,9 +4,11 @@ GH #28991 broke /chat/completions (and /responses) for most models on some
releases: a clean 200 came back but with no real completion. A status check
alone would not have caught it, so each case here asserts the product promise -
a non-empty assistant message and a real model name in the body - across the
three providers wired into the gateway config (OpenAI, Anthropic, Gemini). A
regression that empties the completion for any provider fails that provider's
row here.
providers wired into the gateway config (OpenAI, Anthropic, Gemini, Azure
OpenAI). A regression that empties the completion for any provider fails that
provider's row here. The Azure OpenAI streaming case applies the same standard
to the SSE path: every data event must parse as a chat.completion.chunk and the
deltas must reassemble into real text, not just count as a 200 with chunks.
"""
from __future__ import annotations
@ -15,15 +17,18 @@ import pytest
from e2e_config import unique_marker
from e2e_http import unwrap
from models import ChatBody, ChatMessage
from models import ChatBody, ChatMessage, ChatStreamChunk
from passthrough_client import PassthroughClient
pytestmark = pytest.mark.e2e
AZURE_CHAT_MODEL = "azure-gpt-5.4-mini"
CHAT_MODELS: tuple[tuple[str, str], ...] = (
("gpt-5.5", "openai"),
("claude-haiku-4-5", "anthropic"),
("gemini-2.5-flash", "gemini"),
(AZURE_CHAT_MODEL, "azure_openai"),
)
@ -37,6 +42,7 @@ class TestChatCompletionsRegression:
"llm.chat_completions.openai.basic.nonstream.works",
"llm.chat_completions.anthropic.basic.nonstream.works",
"llm.chat_completions.vertex.basic.nonstream.works",
"llm.chat_completions.azure_openai.basic.nonstream.works",
exercised_on=[],
)
def test_chat_returns_real_completion(
@ -68,3 +74,62 @@ class TestChatCompletionsRegression:
assert (
message is not None and message.content and message.content.strip()
), f"{model} ({route}): 200 with an empty completion (#28991): {response}"
@pytest.mark.covers("llm.chat_completions.azure_openai.basic.stream.works")
def test_azure_openai_stream_returns_real_completion(
self, client: PassthroughClient, scoped_key: str
) -> None:
result = client.gateway.chat_stream(
scoped_key,
ChatBody(
model=AZURE_CHAT_MODEL,
messages=[
ChatMessage(
role="user",
content=f"reply with one word {unique_marker()}",
)
],
max_tokens=512,
stream=True,
),
)
assert result.ok, (
f"{AZURE_CHAT_MODEL}: stream failed with status "
f"{result.status_code}: {result.body[:300]}"
)
assert result.is_streaming, (
f"{AZURE_CHAT_MODEL}: expected text/event-stream, got "
f"{result.content_type}: {result.body[:300]}"
)
assert result.stream_error is None, (
f"{AZURE_CHAT_MODEL}: 200 stream carried an error event: "
f"{result.stream_error}"
)
assert result.events, f"{AZURE_CHAT_MODEL}: stream carried no SSE data events"
assert result.events[-1] == "[DONE]", (
f"{AZURE_CHAT_MODEL}: stream did not terminate with [DONE]: "
f"{result.events[-1][:200]}"
)
chunks = [
ChatStreamChunk.model_validate_json(event) for event in result.events[:-1]
]
assert chunks, f"{AZURE_CHAT_MODEL}: stream held only the [DONE] sentinel"
assert all(
chunk.object == "chat.completion.chunk" for chunk in chunks
), f"{AZURE_CHAT_MODEL}: malformed chunk object types: {result.events[:5]}"
assert any(
chunk.model for chunk in chunks
), f"{AZURE_CHAT_MODEL}: no chunk carried a model name: {result.events[:5]}"
content = "".join(
choice.delta.content or ""
for chunk in chunks
for choice in chunk.choices
if choice.delta is not None
)
assert content.strip(), (
f"{AZURE_CHAT_MODEL}: stream chunks reassembled to an empty "
f"completion (#28991): {result.events[:5]}"
)

View file

@ -191,6 +191,21 @@ class ChatResponse(BaseModel):
service_tier: str | None = None
class ChatStreamDelta(BaseModel):
content: str | None = None
class ChatStreamChoice(BaseModel):
delta: ChatStreamDelta | None = None
class ChatStreamChunk(BaseModel):
id: str | None = None
object: str | None = None
model: str | None = None
choices: list[ChatStreamChoice] = []
class EmbedBody(BaseModel):
model: str
input: str