diff --git a/tests/e2e/docker-compose.yml b/tests/e2e/docker-compose.yml index 4e036b20462..3ba8be37057 100644 --- a/tests/e2e/docker-compose.yml +++ b/tests/e2e/docker-compose.yml @@ -66,9 +66,21 @@ configs: model: openai/text-embedding-3-small api_key: os.environ/OPENAI_API_KEY - - model_name: azure-gpt-5.4-mini + - model_name: azure-${E2E_AZURE_SOL_MODEL:-gpt-5.6-sol} litellm_params: - model: azure/gpt-5.4-mini + model: azure/${E2E_AZURE_SOL_MODEL:-gpt-5.6-sol} + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY + + - model_name: azure-${E2E_AZURE_TERRA_MODEL:-gpt-5.6-terra} + litellm_params: + model: azure/${E2E_AZURE_TERRA_MODEL:-gpt-5.6-terra} + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY + + - model_name: azure-${E2E_AZURE_LUNA_MODEL:-gpt-5.6-luna} + litellm_params: + model: azure/${E2E_AZURE_LUNA_MODEL:-gpt-5.6-luna} api_base: os.environ/AZURE_API_BASE api_key: os.environ/AZURE_API_KEY diff --git a/tests/e2e/e2e_config.py b/tests/e2e/e2e_config.py index 6e6c30709de..6eb040adc9e 100644 --- a/tests/e2e/e2e_config.py +++ b/tests/e2e/e2e_config.py @@ -27,6 +27,15 @@ UI_PASSWORD = os.environ.get("E2E_UI_PASSWORD", MASTER_KEY) CHEAP_ANTHROPIC_MODEL = os.environ.get("E2E_CHEAP_ANTHROPIC_MODEL", "claude-haiku-4-5") CHEAP_OPENAI_MODEL = os.environ.get("E2E_CHEAP_OPENAI_MODEL", "gpt-5.5") +AZURE_CHAT_MODELS = tuple( + f"azure-{os.environ.get(var, default)}" + for var, default in ( + ("E2E_AZURE_SOL_MODEL", "gpt-5.6-sol"), + ("E2E_AZURE_TERRA_MODEL", "gpt-5.6-terra"), + ("E2E_AZURE_LUNA_MODEL", "gpt-5.6-luna"), + ) +) + # Jaeger query API of the compose stack's OTEL trace destination (the `jaeger` # service in docker-compose.yml maps it to host 16686). Trace-completeness tests # read exported spans back through it. diff --git a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py index efcd6fddc2f..1231527eb3a 100644 --- a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py +++ b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py @@ -4,31 +4,31 @@ GH #28991 broke /chat/completions (and /responses) for most models on some releases: a clean 200 came back but with no real completion. A status check alone would not have caught it, so each case here asserts the product promise - a non-empty assistant message and a real model name in the body - across the -providers wired into the gateway config (OpenAI, Anthropic, Gemini, Azure -OpenAI). A regression that empties the completion for any provider fails that -provider's row here. The Azure OpenAI streaming case applies the same standard -to the SSE path: every data event must parse as a chat.completion.chunk and the -deltas must reassemble into real text, not just count as a 200 with chunks. +providers wired into the gateway config (OpenAI, Anthropic, Gemini, and the +Azure OpenAI GPT capability tiers from e2e_config.AZURE_CHAT_MODELS, deployment +names swappable via the E2E_AZURE_*_MODEL environment variables). A regression +that empties the completion for any provider fails that provider's row here. +The Azure OpenAI streaming cases apply the same standard to the SSE path: every +data event must parse as a chat.completion.chunk and the deltas must reassemble +into real text, not just count as a 200 with chunks. """ from __future__ import annotations import pytest -from e2e_config import unique_marker +from e2e_config import AZURE_CHAT_MODELS, unique_marker from e2e_http import unwrap from models import ChatBody, ChatMessage, ChatStreamChunk from passthrough_client import PassthroughClient pytestmark = pytest.mark.e2e -AZURE_CHAT_MODEL = "azure-gpt-5.4-mini" - CHAT_MODELS: tuple[tuple[str, str], ...] = ( ("gpt-5.5", "openai"), ("claude-haiku-4-5", "anthropic"), ("gemini-2.5-flash", "gemini"), - (AZURE_CHAT_MODEL, "azure_openai"), + *((model, "azure_openai") for model in AZURE_CHAT_MODELS), ) @@ -75,14 +75,15 @@ class TestChatCompletionsRegression: message is not None and message.content and message.content.strip() ), f"{model} ({route}): 200 with an empty completion (#28991): {response}" + @pytest.mark.parametrize("model", AZURE_CHAT_MODELS) @pytest.mark.covers("llm.chat_completions.azure_openai.basic.stream.works") def test_azure_openai_stream_returns_real_completion( - self, client: PassthroughClient, scoped_key: str + self, client: PassthroughClient, scoped_key: str, model: str ) -> None: result = client.gateway.chat_stream( scoped_key, ChatBody( - model=AZURE_CHAT_MODEL, + model=model, messages=[ ChatMessage( role="user", @@ -95,33 +96,33 @@ class TestChatCompletionsRegression: ) assert result.ok, ( - f"{AZURE_CHAT_MODEL}: stream failed with status " + f"{model}: stream failed with status " f"{result.status_code}: {result.body[:300]}" ) assert result.is_streaming, ( - f"{AZURE_CHAT_MODEL}: expected text/event-stream, got " + f"{model}: expected text/event-stream, got " f"{result.content_type}: {result.body[:300]}" ) assert result.stream_error is None, ( - f"{AZURE_CHAT_MODEL}: 200 stream carried an error event: " + f"{model}: 200 stream carried an error event: " f"{result.stream_error}" ) - assert result.events, f"{AZURE_CHAT_MODEL}: stream carried no SSE data events" + assert result.events, f"{model}: stream carried no SSE data events" assert result.events[-1] == "[DONE]", ( - f"{AZURE_CHAT_MODEL}: stream did not terminate with [DONE]: " + f"{model}: stream did not terminate with [DONE]: " f"{result.events[-1][:200]}" ) chunks = [ ChatStreamChunk.model_validate_json(event) for event in result.events[:-1] ] - assert chunks, f"{AZURE_CHAT_MODEL}: stream held only the [DONE] sentinel" + assert chunks, f"{model}: stream held only the [DONE] sentinel" assert all( chunk.object == "chat.completion.chunk" for chunk in chunks - ), f"{AZURE_CHAT_MODEL}: malformed chunk object types: {result.events[:5]}" + ), f"{model}: malformed chunk object types: {result.events[:5]}" assert any( chunk.model for chunk in chunks - ), f"{AZURE_CHAT_MODEL}: no chunk carried a model name: {result.events[:5]}" + ), f"{model}: no chunk carried a model name: {result.events[:5]}" content = "".join( choice.delta.content or "" @@ -130,6 +131,6 @@ class TestChatCompletionsRegression: if choice.delta is not None ) assert content.strip(), ( - f"{AZURE_CHAT_MODEL}: stream chunks reassembled to an empty " + f"{model}: stream chunks reassembled to an empty " f"completion (#28991): {result.events[:5]}" )