mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
test(e2e): swap Azure OpenAI coverage to the GPT-5.6 tiers behind env vars
This commit is contained in:
parent
e7c3f7bc75
commit
55f58cde8c
3 changed files with 44 additions and 22 deletions
|
|
@ -66,9 +66,21 @@ configs:
|
|||
model: openai/text-embedding-3-small
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: azure-gpt-5.4-mini
|
||||
- model_name: azure-${E2E_AZURE_SOL_MODEL:-gpt-5.6-sol}
|
||||
litellm_params:
|
||||
model: azure/gpt-5.4-mini
|
||||
model: azure/${E2E_AZURE_SOL_MODEL:-gpt-5.6-sol}
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
||||
- model_name: azure-${E2E_AZURE_TERRA_MODEL:-gpt-5.6-terra}
|
||||
litellm_params:
|
||||
model: azure/${E2E_AZURE_TERRA_MODEL:-gpt-5.6-terra}
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
||||
- model_name: azure-${E2E_AZURE_LUNA_MODEL:-gpt-5.6-luna}
|
||||
litellm_params:
|
||||
model: azure/${E2E_AZURE_LUNA_MODEL:-gpt-5.6-luna}
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
||||
|
|
|
|||
|
|
@ -27,6 +27,15 @@ UI_PASSWORD = os.environ.get("E2E_UI_PASSWORD", MASTER_KEY)
|
|||
CHEAP_ANTHROPIC_MODEL = os.environ.get("E2E_CHEAP_ANTHROPIC_MODEL", "claude-haiku-4-5")
|
||||
CHEAP_OPENAI_MODEL = os.environ.get("E2E_CHEAP_OPENAI_MODEL", "gpt-5.5")
|
||||
|
||||
AZURE_CHAT_MODELS = tuple(
|
||||
f"azure-{os.environ.get(var, default)}"
|
||||
for var, default in (
|
||||
("E2E_AZURE_SOL_MODEL", "gpt-5.6-sol"),
|
||||
("E2E_AZURE_TERRA_MODEL", "gpt-5.6-terra"),
|
||||
("E2E_AZURE_LUNA_MODEL", "gpt-5.6-luna"),
|
||||
)
|
||||
)
|
||||
|
||||
# Jaeger query API of the compose stack's OTEL trace destination (the `jaeger`
|
||||
# service in docker-compose.yml maps it to host 16686). Trace-completeness tests
|
||||
# read exported spans back through it.
|
||||
|
|
|
|||
|
|
@ -4,31 +4,31 @@ GH #28991 broke /chat/completions (and /responses) for most models on some
|
|||
releases: a clean 200 came back but with no real completion. A status check
|
||||
alone would not have caught it, so each case here asserts the product promise -
|
||||
a non-empty assistant message and a real model name in the body - across the
|
||||
providers wired into the gateway config (OpenAI, Anthropic, Gemini, Azure
|
||||
OpenAI). A regression that empties the completion for any provider fails that
|
||||
provider's row here. The Azure OpenAI streaming case applies the same standard
|
||||
to the SSE path: every data event must parse as a chat.completion.chunk and the
|
||||
deltas must reassemble into real text, not just count as a 200 with chunks.
|
||||
providers wired into the gateway config (OpenAI, Anthropic, Gemini, and the
|
||||
Azure OpenAI GPT capability tiers from e2e_config.AZURE_CHAT_MODELS, deployment
|
||||
names swappable via the E2E_AZURE_*_MODEL environment variables). A regression
|
||||
that empties the completion for any provider fails that provider's row here.
|
||||
The Azure OpenAI streaming cases apply the same standard to the SSE path: every
|
||||
data event must parse as a chat.completion.chunk and the deltas must reassemble
|
||||
into real text, not just count as a 200 with chunks.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_config import AZURE_CHAT_MODELS, unique_marker
|
||||
from e2e_http import unwrap
|
||||
from models import ChatBody, ChatMessage, ChatStreamChunk
|
||||
from passthrough_client import PassthroughClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
AZURE_CHAT_MODEL = "azure-gpt-5.4-mini"
|
||||
|
||||
CHAT_MODELS: tuple[tuple[str, str], ...] = (
|
||||
("gpt-5.5", "openai"),
|
||||
("claude-haiku-4-5", "anthropic"),
|
||||
("gemini-2.5-flash", "gemini"),
|
||||
(AZURE_CHAT_MODEL, "azure_openai"),
|
||||
*((model, "azure_openai") for model in AZURE_CHAT_MODELS),
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -75,14 +75,15 @@ class TestChatCompletionsRegression:
|
|||
message is not None and message.content and message.content.strip()
|
||||
), f"{model} ({route}): 200 with an empty completion (#28991): {response}"
|
||||
|
||||
@pytest.mark.parametrize("model", AZURE_CHAT_MODELS)
|
||||
@pytest.mark.covers("llm.chat_completions.azure_openai.basic.stream.works")
|
||||
def test_azure_openai_stream_returns_real_completion(
|
||||
self, client: PassthroughClient, scoped_key: str
|
||||
self, client: PassthroughClient, scoped_key: str, model: str
|
||||
) -> None:
|
||||
result = client.gateway.chat_stream(
|
||||
scoped_key,
|
||||
ChatBody(
|
||||
model=AZURE_CHAT_MODEL,
|
||||
model=model,
|
||||
messages=[
|
||||
ChatMessage(
|
||||
role="user",
|
||||
|
|
@ -95,33 +96,33 @@ class TestChatCompletionsRegression:
|
|||
)
|
||||
|
||||
assert result.ok, (
|
||||
f"{AZURE_CHAT_MODEL}: stream failed with status "
|
||||
f"{model}: stream failed with status "
|
||||
f"{result.status_code}: {result.body[:300]}"
|
||||
)
|
||||
assert result.is_streaming, (
|
||||
f"{AZURE_CHAT_MODEL}: expected text/event-stream, got "
|
||||
f"{model}: expected text/event-stream, got "
|
||||
f"{result.content_type}: {result.body[:300]}"
|
||||
)
|
||||
assert result.stream_error is None, (
|
||||
f"{AZURE_CHAT_MODEL}: 200 stream carried an error event: "
|
||||
f"{model}: 200 stream carried an error event: "
|
||||
f"{result.stream_error}"
|
||||
)
|
||||
assert result.events, f"{AZURE_CHAT_MODEL}: stream carried no SSE data events"
|
||||
assert result.events, f"{model}: stream carried no SSE data events"
|
||||
assert result.events[-1] == "[DONE]", (
|
||||
f"{AZURE_CHAT_MODEL}: stream did not terminate with [DONE]: "
|
||||
f"{model}: stream did not terminate with [DONE]: "
|
||||
f"{result.events[-1][:200]}"
|
||||
)
|
||||
|
||||
chunks = [
|
||||
ChatStreamChunk.model_validate_json(event) for event in result.events[:-1]
|
||||
]
|
||||
assert chunks, f"{AZURE_CHAT_MODEL}: stream held only the [DONE] sentinel"
|
||||
assert chunks, f"{model}: stream held only the [DONE] sentinel"
|
||||
assert all(
|
||||
chunk.object == "chat.completion.chunk" for chunk in chunks
|
||||
), f"{AZURE_CHAT_MODEL}: malformed chunk object types: {result.events[:5]}"
|
||||
), f"{model}: malformed chunk object types: {result.events[:5]}"
|
||||
assert any(
|
||||
chunk.model for chunk in chunks
|
||||
), f"{AZURE_CHAT_MODEL}: no chunk carried a model name: {result.events[:5]}"
|
||||
), f"{model}: no chunk carried a model name: {result.events[:5]}"
|
||||
|
||||
content = "".join(
|
||||
choice.delta.content or ""
|
||||
|
|
@ -130,6 +131,6 @@ class TestChatCompletionsRegression:
|
|||
if choice.delta is not None
|
||||
)
|
||||
assert content.strip(), (
|
||||
f"{AZURE_CHAT_MODEL}: stream chunks reassembled to an empty "
|
||||
f"{model}: stream chunks reassembled to an empty "
|
||||
f"completion (#28991): {result.events[:5]}"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue