test(e2e): guard Azure reasoning_effort=none via base_model and xfail the dual token-param bug

Adds two customer-regression rows to the Azure chat e2e suite. The first
sends reasoning_effort=none to a custom-named deployment (gpt-5.6-sol-e2e,
swappable via E2E_AZURE_CUSTOM_MODEL) whose capabilities resolve through
base_model, asserting the request completes with zero reasoning tokens on
a prompt that reasons at default effort, so both a gate 400 (GH #31243,
SDK fix in PR #28490) and a silently dropped param fail the row. The
second, a strict xfail until GH #31614 is fixed, sends a client
max_completion_tokens to a gpt-4o deployment carrying a config-level
max_tokens default; the proxy forwards both and Azure rejects the pair
This commit is contained in:
mateo-berri 2026-07-16 15:41:00 -07:00
parent 294e01fa3d
commit 187cab5049
5 changed files with 132 additions and 1 deletions

View file

@ -31,6 +31,8 @@
- {id: llm.chat_completions.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Azure OpenAI deployments"}
- {id: llm.chat_completions.azure_openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Azure OpenAI"}
- {id: llm.chat_completions.azure_openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Azure OpenAI function_calling"}
- {id: llm.chat_completions.azure_openai.thinking.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: thinking, streaming: nonstream, assertions: [works], source: "llms/azure/chat/gpt_5_transformation.py", rationale: "reasoning_effort=none on a custom-named deployment must resolve capabilities via base_model and reach Azure with reasoning disabled (GH #31243; SDK gate fixed by PR #28490, proxy base_model registration is the second guard; test fails if either layer 400s or silently drops the param)"}
- {id: llm.chat_completions.azure_openai.basic.nonstream.token_param_dedup, module: llm, tier: P1, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/azure/chat/transformation.py", rationale: "Config-level max_tokens default plus client max_completion_tokens must not forward both to Azure, which 400s on the pair; open bug GH #31614, covering test is xfail until a fix lands", fail_before_fix: proven}
- {id: llm.chat_completions.azure_foundry.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: azure_foundry, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Azure Foundry (azure_ai); newer, smoke"}
- {id: llm.messages.anthropic.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Core endpoint; Anthropic Messages native"}
- {id: llm.messages.anthropic.basic.stream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: stream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Streaming Messages API"}

View file

@ -84,6 +84,22 @@ configs:
api_base: os.environ/AZURE_API_BASE
api_key: os.environ/AZURE_API_KEY
- model_name: azure-${E2E_AZURE_CUSTOM_MODEL:-gpt-5.6-sol-e2e}
litellm_params:
model: azure/${E2E_AZURE_CUSTOM_MODEL:-gpt-5.6-sol-e2e}
api_base: os.environ/AZURE_API_BASE
api_key: os.environ/AZURE_API_KEY
drop_params: false
model_info:
base_model: ${E2E_AZURE_CUSTOM_BASE_MODEL:-azure/gpt-5.6-sol}
- model_name: azure-${E2E_AZURE_GPT4O_MODEL:-gpt-4o}
litellm_params:
model: azure/${E2E_AZURE_GPT4O_MODEL:-gpt-4o}
api_base: os.environ/AZURE_API_BASE
api_key: os.environ/AZURE_API_KEY
max_tokens: 512
services:
litellm:
image: ghcr.io/berriai/litellm:main-latest

View file

@ -36,6 +36,12 @@ AZURE_CHAT_MODELS = tuple(
)
)
AZURE_CUSTOM_NAME_CHAT_MODEL = (
f"azure-{os.environ.get('E2E_AZURE_CUSTOM_MODEL', 'gpt-5.6-sol-e2e')}"
)
AZURE_GPT4O_CHAT_MODEL = f"azure-{os.environ.get('E2E_AZURE_GPT4O_MODEL', 'gpt-4o')}"
# Jaeger query API of the compose stack's OTEL trace destination (the `jaeger`
# service in docker-compose.yml maps it to host 16686). Trace-completeness tests
# read exported spans back through it.

View file

@ -11,13 +11,28 @@ that empties the completion for any provider fails that provider's row here.
The Azure OpenAI streaming cases apply the same standard to the SSE path: every
data event must parse as a chat.completion.chunk and the deltas must reassemble
into real text, not just count as a 200 with chunks.
Two cases guard specific customer-reported Azure regressions beyond the happy
path. GH #31243: reasoning_effort='none' against a custom-named deployment must
resolve model capabilities through base_model and reach Azure with reasoning
actually disabled; the prompt is chosen to spend reasoning tokens at default
effort, so the test fails on a gate 400 (the SDK-level bug PR #28490 fixed) and
also on a silently dropped param (which drop_params=true would otherwise mask).
GH #31614 (still open, marked xfail): a config-level max_tokens default
combined with a client-sent max_completion_tokens forwards both parameters to
Azure, which rejects the pair; the strict xfail flips when a fix lands.
"""
from __future__ import annotations
import pytest
from e2e_config import AZURE_CHAT_MODELS, unique_marker
from e2e_config import (
AZURE_CHAT_MODELS,
AZURE_CUSTOM_NAME_CHAT_MODEL,
AZURE_GPT4O_CHAT_MODEL,
unique_marker,
)
from e2e_http import unwrap
from models import ChatBody, ChatMessage, ChatStreamChunk
from passthrough_client import PassthroughClient
@ -134,3 +149,89 @@ class TestChatCompletionsRegression:
f"{model}: stream chunks reassembled to an empty "
f"completion (#28991): {result.events[:5]}"
)
@pytest.mark.covers("llm.chat_completions.azure_openai.thinking.nonstream.works")
def test_azure_custom_deployment_name_reasoning_effort_none(
self, client: PassthroughClient, scoped_key: str
) -> None:
response = unwrap(
client.gateway.chat(
scoped_key,
ChatBody(
model=AZURE_CUSTOM_NAME_CHAT_MODEL,
messages=[
ChatMessage(
role="user",
content=(
"A farmer has 17 sheep, all but 9 run away, then "
"he buys twice as many as remain minus 3. How many "
"sheep? Reply with just the number. "
f"(session {unique_marker()})"
),
)
],
max_completion_tokens=2000,
reasoning_effort="none",
),
)
)
assert response.choices, (
f"{AZURE_CUSTOM_NAME_CHAT_MODEL}: reasoning_effort='none' on a "
f"custom-named deployment must resolve capabilities via base_model "
f"(GH #31243): {response}"
)
message = response.choices[0].message
assert message is not None and message.content and message.content.strip(), (
f"{AZURE_CUSTOM_NAME_CHAT_MODEL}: empty completion for "
f"reasoning_effort='none' (GH #31243): {response}"
)
reasoning_tokens = (
response.usage.completion_tokens_details.reasoning_tokens
if response.usage and response.usage.completion_tokens_details
else None
)
assert not reasoning_tokens, (
f"{AZURE_CUSTOM_NAME_CHAT_MODEL}: reasoning_effort='none' must "
f"disable reasoning, but the model spent {reasoning_tokens} "
f"reasoning tokens (GH #31243): {response.usage}"
)
@pytest.mark.covers(
"llm.chat_completions.azure_openai.basic.nonstream.token_param_dedup"
)
@pytest.mark.xfail(
strict=True,
reason=(
"GH #31614: a config-level max_tokens default plus a client "
"max_completion_tokens forwards both to Azure, which rejects the pair"
),
)
def test_azure_config_token_cap_with_client_max_completion_tokens(
self, client: PassthroughClient, scoped_key: str
) -> None:
response = unwrap(
client.gateway.chat(
scoped_key,
ChatBody(
model=AZURE_GPT4O_CHAT_MODEL,
messages=[
ChatMessage(
role="user",
content=f"reply with one word {unique_marker()}",
)
],
max_completion_tokens=256,
),
)
)
assert response.choices, (
f"{AZURE_GPT4O_CHAT_MODEL}: client max_completion_tokens on a "
f"deployment with a config max_tokens default must still complete "
f"(GH #31614): {response}"
)
message = response.choices[0].message
assert message is not None and message.content and message.content.strip(), (
f"{AZURE_GPT4O_CHAT_MODEL}: empty completion (GH #31614): {response}"
)

View file

@ -140,6 +140,7 @@ class ChatBody(BaseModel):
messages: list[ChatMessage]
stream: bool = False
max_tokens: int | None = None
max_completion_tokens: int | None = None
user: str | None = None
metadata: ChatMetadata | None = None
reasoning_effort: str | None = None
@ -174,6 +175,10 @@ class PromptTokensDetails(BaseModel):
cached_tokens: int | None = None
class CompletionTokensDetails(BaseModel):
reasoning_tokens: int | None = None
class Usage(BaseModel):
prompt_tokens: int | None = None
completion_tokens: int | None = None
@ -181,6 +186,7 @@ class Usage(BaseModel):
cache_read_input_tokens: int | None = None
cache_creation_input_tokens: int | None = None
prompt_tokens_details: PromptTokensDetails | None = None
completion_tokens_details: CompletionTokensDetails | None = None
class ChatResponse(BaseModel):