diff --git a/tests/e2e/coverage_registry/llm_conversational.yaml b/tests/e2e/coverage_registry/llm_conversational.yaml index 40d766e7ce5..1788dc411fa 100644 --- a/tests/e2e/coverage_registry/llm_conversational.yaml +++ b/tests/e2e/coverage_registry/llm_conversational.yaml @@ -31,6 +31,8 @@ - {id: llm.chat_completions.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Azure OpenAI deployments"} - {id: llm.chat_completions.azure_openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Azure OpenAI"} - {id: llm.chat_completions.azure_openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Azure OpenAI function_calling"} +- {id: llm.chat_completions.azure_openai.thinking.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: thinking, streaming: nonstream, assertions: [works], source: "llms/azure/chat/gpt_5_transformation.py", rationale: "reasoning_effort=none on a custom-named deployment must resolve capabilities via base_model and reach Azure with reasoning disabled (GH #31243; SDK gate fixed by PR #28490, proxy base_model registration is the second guard; test fails if either layer 400s or silently drops the param)"} +- {id: llm.chat_completions.azure_openai.basic.nonstream.token_param_dedup, module: llm, tier: P1, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/azure/chat/transformation.py", rationale: "Config-level max_tokens default plus client max_completion_tokens must not forward both to Azure, which 400s on the pair; open bug GH #31614, covering test is xfail until a fix lands", fail_before_fix: proven} - {id: llm.chat_completions.azure_foundry.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: azure_foundry, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Azure Foundry (azure_ai); newer, smoke"} - {id: llm.messages.anthropic.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Core endpoint; Anthropic Messages native"} - {id: llm.messages.anthropic.basic.stream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: stream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Streaming Messages API"} diff --git a/tests/e2e/docker-compose.yml b/tests/e2e/docker-compose.yml index 3ba8be37057..7af10c91fb7 100644 --- a/tests/e2e/docker-compose.yml +++ b/tests/e2e/docker-compose.yml @@ -84,6 +84,22 @@ configs: api_base: os.environ/AZURE_API_BASE api_key: os.environ/AZURE_API_KEY + - model_name: azure-${E2E_AZURE_CUSTOM_MODEL:-gpt-5.6-sol-e2e} + litellm_params: + model: azure/${E2E_AZURE_CUSTOM_MODEL:-gpt-5.6-sol-e2e} + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY + drop_params: false + model_info: + base_model: ${E2E_AZURE_CUSTOM_BASE_MODEL:-azure/gpt-5.6-sol} + + - model_name: azure-${E2E_AZURE_GPT4O_MODEL:-gpt-4o} + litellm_params: + model: azure/${E2E_AZURE_GPT4O_MODEL:-gpt-4o} + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY + max_tokens: 512 + services: litellm: image: ghcr.io/berriai/litellm:main-latest diff --git a/tests/e2e/e2e_config.py b/tests/e2e/e2e_config.py index 6eb040adc9e..4d2f8fa36e4 100644 --- a/tests/e2e/e2e_config.py +++ b/tests/e2e/e2e_config.py @@ -36,6 +36,12 @@ AZURE_CHAT_MODELS = tuple( ) ) +AZURE_CUSTOM_NAME_CHAT_MODEL = ( + f"azure-{os.environ.get('E2E_AZURE_CUSTOM_MODEL', 'gpt-5.6-sol-e2e')}" +) + +AZURE_GPT4O_CHAT_MODEL = f"azure-{os.environ.get('E2E_AZURE_GPT4O_MODEL', 'gpt-4o')}" + # Jaeger query API of the compose stack's OTEL trace destination (the `jaeger` # service in docker-compose.yml maps it to host 16686). Trace-completeness tests # read exported spans back through it. diff --git a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py index 1231527eb3a..62a29b46942 100644 --- a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py +++ b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py @@ -11,13 +11,28 @@ that empties the completion for any provider fails that provider's row here. The Azure OpenAI streaming cases apply the same standard to the SSE path: every data event must parse as a chat.completion.chunk and the deltas must reassemble into real text, not just count as a 200 with chunks. + +Two cases guard specific customer-reported Azure regressions beyond the happy +path. GH #31243: reasoning_effort='none' against a custom-named deployment must +resolve model capabilities through base_model and reach Azure with reasoning +actually disabled; the prompt is chosen to spend reasoning tokens at default +effort, so the test fails on a gate 400 (the SDK-level bug PR #28490 fixed) and +also on a silently dropped param (which drop_params=true would otherwise mask). +GH #31614 (still open, marked xfail): a config-level max_tokens default +combined with a client-sent max_completion_tokens forwards both parameters to +Azure, which rejects the pair; the strict xfail flips when a fix lands. """ from __future__ import annotations import pytest -from e2e_config import AZURE_CHAT_MODELS, unique_marker +from e2e_config import ( + AZURE_CHAT_MODELS, + AZURE_CUSTOM_NAME_CHAT_MODEL, + AZURE_GPT4O_CHAT_MODEL, + unique_marker, +) from e2e_http import unwrap from models import ChatBody, ChatMessage, ChatStreamChunk from passthrough_client import PassthroughClient @@ -134,3 +149,89 @@ class TestChatCompletionsRegression: f"{model}: stream chunks reassembled to an empty " f"completion (#28991): {result.events[:5]}" ) + + @pytest.mark.covers("llm.chat_completions.azure_openai.thinking.nonstream.works") + def test_azure_custom_deployment_name_reasoning_effort_none( + self, client: PassthroughClient, scoped_key: str + ) -> None: + response = unwrap( + client.gateway.chat( + scoped_key, + ChatBody( + model=AZURE_CUSTOM_NAME_CHAT_MODEL, + messages=[ + ChatMessage( + role="user", + content=( + "A farmer has 17 sheep, all but 9 run away, then " + "he buys twice as many as remain minus 3. How many " + "sheep? Reply with just the number. " + f"(session {unique_marker()})" + ), + ) + ], + max_completion_tokens=2000, + reasoning_effort="none", + ), + ) + ) + + assert response.choices, ( + f"{AZURE_CUSTOM_NAME_CHAT_MODEL}: reasoning_effort='none' on a " + f"custom-named deployment must resolve capabilities via base_model " + f"(GH #31243): {response}" + ) + message = response.choices[0].message + assert message is not None and message.content and message.content.strip(), ( + f"{AZURE_CUSTOM_NAME_CHAT_MODEL}: empty completion for " + f"reasoning_effort='none' (GH #31243): {response}" + ) + reasoning_tokens = ( + response.usage.completion_tokens_details.reasoning_tokens + if response.usage and response.usage.completion_tokens_details + else None + ) + assert not reasoning_tokens, ( + f"{AZURE_CUSTOM_NAME_CHAT_MODEL}: reasoning_effort='none' must " + f"disable reasoning, but the model spent {reasoning_tokens} " + f"reasoning tokens (GH #31243): {response.usage}" + ) + + @pytest.mark.covers( + "llm.chat_completions.azure_openai.basic.nonstream.token_param_dedup" + ) + @pytest.mark.xfail( + strict=True, + reason=( + "GH #31614: a config-level max_tokens default plus a client " + "max_completion_tokens forwards both to Azure, which rejects the pair" + ), + ) + def test_azure_config_token_cap_with_client_max_completion_tokens( + self, client: PassthroughClient, scoped_key: str + ) -> None: + response = unwrap( + client.gateway.chat( + scoped_key, + ChatBody( + model=AZURE_GPT4O_CHAT_MODEL, + messages=[ + ChatMessage( + role="user", + content=f"reply with one word {unique_marker()}", + ) + ], + max_completion_tokens=256, + ), + ) + ) + + assert response.choices, ( + f"{AZURE_GPT4O_CHAT_MODEL}: client max_completion_tokens on a " + f"deployment with a config max_tokens default must still complete " + f"(GH #31614): {response}" + ) + message = response.choices[0].message + assert message is not None and message.content and message.content.strip(), ( + f"{AZURE_GPT4O_CHAT_MODEL}: empty completion (GH #31614): {response}" + ) diff --git a/tests/e2e/models.py b/tests/e2e/models.py index 6e3bd8ad554..93a78ea3873 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -140,6 +140,7 @@ class ChatBody(BaseModel): messages: list[ChatMessage] stream: bool = False max_tokens: int | None = None + max_completion_tokens: int | None = None user: str | None = None metadata: ChatMetadata | None = None reasoning_effort: str | None = None @@ -174,6 +175,10 @@ class PromptTokensDetails(BaseModel): cached_tokens: int | None = None +class CompletionTokensDetails(BaseModel): + reasoning_tokens: int | None = None + + class Usage(BaseModel): prompt_tokens: int | None = None completion_tokens: int | None = None @@ -181,6 +186,7 @@ class Usage(BaseModel): cache_read_input_tokens: int | None = None cache_creation_input_tokens: int | None = None prompt_tokens_details: PromptTokensDetails | None = None + completion_tokens_details: CompletionTokensDetails | None = None class ChatResponse(BaseModel):