test(e2e): swap Azure OpenAI coverage to the GPT-5.6 tiers behind env vars

This commit is contained in:
mateo-berri 2026-07-15 16:26:43 -07:00
parent e7c3f7bc75
commit 55f58cde8c
3 changed files with 44 additions and 22 deletions

View file

@ -66,9 +66,21 @@ configs:
model: openai/text-embedding-3-small
api_key: os.environ/OPENAI_API_KEY
- model_name: azure-gpt-5.4-mini
- model_name: azure-${E2E_AZURE_SOL_MODEL:-gpt-5.6-sol}
litellm_params:
model: azure/gpt-5.4-mini
model: azure/${E2E_AZURE_SOL_MODEL:-gpt-5.6-sol}
api_base: os.environ/AZURE_API_BASE
api_key: os.environ/AZURE_API_KEY
- model_name: azure-${E2E_AZURE_TERRA_MODEL:-gpt-5.6-terra}
litellm_params:
model: azure/${E2E_AZURE_TERRA_MODEL:-gpt-5.6-terra}
api_base: os.environ/AZURE_API_BASE
api_key: os.environ/AZURE_API_KEY
- model_name: azure-${E2E_AZURE_LUNA_MODEL:-gpt-5.6-luna}
litellm_params:
model: azure/${E2E_AZURE_LUNA_MODEL:-gpt-5.6-luna}
api_base: os.environ/AZURE_API_BASE
api_key: os.environ/AZURE_API_KEY

View file

@ -27,6 +27,15 @@ UI_PASSWORD = os.environ.get("E2E_UI_PASSWORD", MASTER_KEY)
CHEAP_ANTHROPIC_MODEL = os.environ.get("E2E_CHEAP_ANTHROPIC_MODEL", "claude-haiku-4-5")
CHEAP_OPENAI_MODEL = os.environ.get("E2E_CHEAP_OPENAI_MODEL", "gpt-5.5")
AZURE_CHAT_MODELS = tuple(
f"azure-{os.environ.get(var, default)}"
for var, default in (
("E2E_AZURE_SOL_MODEL", "gpt-5.6-sol"),
("E2E_AZURE_TERRA_MODEL", "gpt-5.6-terra"),
("E2E_AZURE_LUNA_MODEL", "gpt-5.6-luna"),
)
)
# Jaeger query API of the compose stack's OTEL trace destination (the `jaeger`
# service in docker-compose.yml maps it to host 16686). Trace-completeness tests
# read exported spans back through it.

View file

@ -4,31 +4,31 @@ GH #28991 broke /chat/completions (and /responses) for most models on some
releases: a clean 200 came back but with no real completion. A status check
alone would not have caught it, so each case here asserts the product promise -
a non-empty assistant message and a real model name in the body - across the
providers wired into the gateway config (OpenAI, Anthropic, Gemini, Azure
OpenAI). A regression that empties the completion for any provider fails that
provider's row here. The Azure OpenAI streaming case applies the same standard
to the SSE path: every data event must parse as a chat.completion.chunk and the
deltas must reassemble into real text, not just count as a 200 with chunks.
providers wired into the gateway config (OpenAI, Anthropic, Gemini, and the
Azure OpenAI GPT capability tiers from e2e_config.AZURE_CHAT_MODELS, deployment
names swappable via the E2E_AZURE_*_MODEL environment variables). A regression
that empties the completion for any provider fails that provider's row here.
The Azure OpenAI streaming cases apply the same standard to the SSE path: every
data event must parse as a chat.completion.chunk and the deltas must reassemble
into real text, not just count as a 200 with chunks.
"""
from __future__ import annotations
import pytest
from e2e_config import unique_marker
from e2e_config import AZURE_CHAT_MODELS, unique_marker
from e2e_http import unwrap
from models import ChatBody, ChatMessage, ChatStreamChunk
from passthrough_client import PassthroughClient
pytestmark = pytest.mark.e2e
AZURE_CHAT_MODEL = "azure-gpt-5.4-mini"
CHAT_MODELS: tuple[tuple[str, str], ...] = (
("gpt-5.5", "openai"),
("claude-haiku-4-5", "anthropic"),
("gemini-2.5-flash", "gemini"),
(AZURE_CHAT_MODEL, "azure_openai"),
*((model, "azure_openai") for model in AZURE_CHAT_MODELS),
)
@ -75,14 +75,15 @@ class TestChatCompletionsRegression:
message is not None and message.content and message.content.strip()
), f"{model} ({route}): 200 with an empty completion (#28991): {response}"
@pytest.mark.parametrize("model", AZURE_CHAT_MODELS)
@pytest.mark.covers("llm.chat_completions.azure_openai.basic.stream.works")
def test_azure_openai_stream_returns_real_completion(
self, client: PassthroughClient, scoped_key: str
self, client: PassthroughClient, scoped_key: str, model: str
) -> None:
result = client.gateway.chat_stream(
scoped_key,
ChatBody(
model=AZURE_CHAT_MODEL,
model=model,
messages=[
ChatMessage(
role="user",
@ -95,33 +96,33 @@ class TestChatCompletionsRegression:
)
assert result.ok, (
f"{AZURE_CHAT_MODEL}: stream failed with status "
f"{model}: stream failed with status "
f"{result.status_code}: {result.body[:300]}"
)
assert result.is_streaming, (
f"{AZURE_CHAT_MODEL}: expected text/event-stream, got "
f"{model}: expected text/event-stream, got "
f"{result.content_type}: {result.body[:300]}"
)
assert result.stream_error is None, (
f"{AZURE_CHAT_MODEL}: 200 stream carried an error event: "
f"{model}: 200 stream carried an error event: "
f"{result.stream_error}"
)
assert result.events, f"{AZURE_CHAT_MODEL}: stream carried no SSE data events"
assert result.events, f"{model}: stream carried no SSE data events"
assert result.events[-1] == "[DONE]", (
f"{AZURE_CHAT_MODEL}: stream did not terminate with [DONE]: "
f"{model}: stream did not terminate with [DONE]: "
f"{result.events[-1][:200]}"
)
chunks = [
ChatStreamChunk.model_validate_json(event) for event in result.events[:-1]
]
assert chunks, f"{AZURE_CHAT_MODEL}: stream held only the [DONE] sentinel"
assert chunks, f"{model}: stream held only the [DONE] sentinel"
assert all(
chunk.object == "chat.completion.chunk" for chunk in chunks
), f"{AZURE_CHAT_MODEL}: malformed chunk object types: {result.events[:5]}"
), f"{model}: malformed chunk object types: {result.events[:5]}"
assert any(
chunk.model for chunk in chunks
), f"{AZURE_CHAT_MODEL}: no chunk carried a model name: {result.events[:5]}"
), f"{model}: no chunk carried a model name: {result.events[:5]}"
content = "".join(
choice.delta.content or ""
@ -130,6 +131,6 @@ class TestChatCompletionsRegression:
if choice.delta is not None
)
assert content.strip(), (
f"{AZURE_CHAT_MODEL}: stream chunks reassembled to an empty "
f"{model}: stream chunks reassembled to an empty "
f"completion (#28991): {result.events[:5]}"
)