mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
test(e2e): cover /chat/completions on Azure OpenAI, streaming and non-streaming
This commit is contained in:
parent
c6d49a85b2
commit
e7c3f7bc75
5 changed files with 114 additions and 19 deletions
|
|
@ -29,6 +29,7 @@
|
|||
- {id: llm.chat_completions.vertex.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Gemini vision"}
|
||||
- {id: llm.chat_completions.vertex.prompt_cache_5m.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: vertex, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Vertex Gemini prompt caching"}
|
||||
- {id: llm.chat_completions.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Azure OpenAI deployments"}
|
||||
- {id: llm.chat_completions.azure_openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Azure OpenAI"}
|
||||
- {id: llm.chat_completions.azure_openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: azure_openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Azure OpenAI function_calling"}
|
||||
- {id: llm.chat_completions.azure_foundry.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: azure_foundry, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Azure Foundry (azure_ai); newer, smoke"}
|
||||
- {id: llm.messages.anthropic.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "anthropic_endpoints/endpoints.py:64", rationale: "Core endpoint; Anthropic Messages native"}
|
||||
|
|
|
|||
|
|
@ -66,6 +66,12 @@ configs:
|
|||
model: openai/text-embedding-3-small
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: azure-gpt-5.4-mini
|
||||
litellm_params:
|
||||
model: azure/gpt-5.4-mini
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
|
||||
services:
|
||||
litellm:
|
||||
image: ghcr.io/berriai/litellm:main-latest
|
||||
|
|
|
|||
|
|
@ -121,6 +121,7 @@ class StreamingResponse(BaseModel):
|
|||
headers: dict[str, str] = {}
|
||||
body: str
|
||||
chunks: int = 0 # streamed events (0 for non-streaming)
|
||||
events: tuple[str, ...] = () # SSE "data:" payloads, prefix stripped (streaming only)
|
||||
# First in-stream error event, if any. A streamed call commits its HTTP 200
|
||||
# before the upstream completes, so upstream failures (e.g. insufficient
|
||||
# quota) arrive as SSE error events inside an otherwise-successful response;
|
||||
|
|
@ -293,20 +294,22 @@ def _streaming_outcome(resp: requests.Response, stream: bool) -> StreamingRespon
|
|||
headers=headers,
|
||||
body=resp.text,
|
||||
)
|
||||
lines = cast("Iterator[bytes]", resp.iter_lines())
|
||||
chunks = 0
|
||||
stream_error: str | None = None
|
||||
for line in lines:
|
||||
if not line:
|
||||
continue
|
||||
chunks += 1
|
||||
if stream_error is None and (
|
||||
line.startswith(b"event: error")
|
||||
or b'"type":"error"' in line
|
||||
or b'"type": "error"' in line
|
||||
or line.startswith(b'data: {"error"')
|
||||
):
|
||||
stream_error = line.decode(errors="replace")[:300]
|
||||
raw_lines = tuple(
|
||||
line.decode(errors="replace")
|
||||
for line in cast("Iterator[bytes]", resp.iter_lines())
|
||||
if line
|
||||
)
|
||||
stream_error = next(
|
||||
(
|
||||
line[:300]
|
||||
for line in raw_lines
|
||||
if line.startswith("event: error")
|
||||
or '"type":"error"' in line
|
||||
or '"type": "error"' in line
|
||||
or line.startswith('data: {"error"')
|
||||
),
|
||||
None,
|
||||
)
|
||||
return StreamingResponse(
|
||||
status_code=resp.status_code,
|
||||
call_id=call_id,
|
||||
|
|
@ -314,7 +317,12 @@ def _streaming_outcome(resp: requests.Response, stream: bool) -> StreamingRespon
|
|||
content_type=content_type,
|
||||
headers=headers,
|
||||
body="<streamed>",
|
||||
chunks=chunks,
|
||||
chunks=len(raw_lines),
|
||||
events=tuple(
|
||||
line.removeprefix("data:").strip()
|
||||
for line in raw_lines
|
||||
if line.startswith("data:")
|
||||
),
|
||||
stream_error=stream_error,
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -4,9 +4,11 @@ GH #28991 broke /chat/completions (and /responses) for most models on some
|
|||
releases: a clean 200 came back but with no real completion. A status check
|
||||
alone would not have caught it, so each case here asserts the product promise -
|
||||
a non-empty assistant message and a real model name in the body - across the
|
||||
three providers wired into the gateway config (OpenAI, Anthropic, Gemini). A
|
||||
regression that empties the completion for any provider fails that provider's
|
||||
row here.
|
||||
providers wired into the gateway config (OpenAI, Anthropic, Gemini, Azure
|
||||
OpenAI). A regression that empties the completion for any provider fails that
|
||||
provider's row here. The Azure OpenAI streaming case applies the same standard
|
||||
to the SSE path: every data event must parse as a chat.completion.chunk and the
|
||||
deltas must reassemble into real text, not just count as a 200 with chunks.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -15,15 +17,18 @@ import pytest
|
|||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from models import ChatBody, ChatMessage
|
||||
from models import ChatBody, ChatMessage, ChatStreamChunk
|
||||
from passthrough_client import PassthroughClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
AZURE_CHAT_MODEL = "azure-gpt-5.4-mini"
|
||||
|
||||
CHAT_MODELS: tuple[tuple[str, str], ...] = (
|
||||
("gpt-5.5", "openai"),
|
||||
("claude-haiku-4-5", "anthropic"),
|
||||
("gemini-2.5-flash", "gemini"),
|
||||
(AZURE_CHAT_MODEL, "azure_openai"),
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -37,6 +42,7 @@ class TestChatCompletionsRegression:
|
|||
"llm.chat_completions.openai.basic.nonstream.works",
|
||||
"llm.chat_completions.anthropic.basic.nonstream.works",
|
||||
"llm.chat_completions.vertex.basic.nonstream.works",
|
||||
"llm.chat_completions.azure_openai.basic.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
def test_chat_returns_real_completion(
|
||||
|
|
@ -68,3 +74,62 @@ class TestChatCompletionsRegression:
|
|||
assert (
|
||||
message is not None and message.content and message.content.strip()
|
||||
), f"{model} ({route}): 200 with an empty completion (#28991): {response}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.azure_openai.basic.stream.works")
|
||||
def test_azure_openai_stream_returns_real_completion(
|
||||
self, client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.gateway.chat_stream(
|
||||
scoped_key,
|
||||
ChatBody(
|
||||
model=AZURE_CHAT_MODEL,
|
||||
messages=[
|
||||
ChatMessage(
|
||||
role="user",
|
||||
content=f"reply with one word {unique_marker()}",
|
||||
)
|
||||
],
|
||||
max_tokens=512,
|
||||
stream=True,
|
||||
),
|
||||
)
|
||||
|
||||
assert result.ok, (
|
||||
f"{AZURE_CHAT_MODEL}: stream failed with status "
|
||||
f"{result.status_code}: {result.body[:300]}"
|
||||
)
|
||||
assert result.is_streaming, (
|
||||
f"{AZURE_CHAT_MODEL}: expected text/event-stream, got "
|
||||
f"{result.content_type}: {result.body[:300]}"
|
||||
)
|
||||
assert result.stream_error is None, (
|
||||
f"{AZURE_CHAT_MODEL}: 200 stream carried an error event: "
|
||||
f"{result.stream_error}"
|
||||
)
|
||||
assert result.events, f"{AZURE_CHAT_MODEL}: stream carried no SSE data events"
|
||||
assert result.events[-1] == "[DONE]", (
|
||||
f"{AZURE_CHAT_MODEL}: stream did not terminate with [DONE]: "
|
||||
f"{result.events[-1][:200]}"
|
||||
)
|
||||
|
||||
chunks = [
|
||||
ChatStreamChunk.model_validate_json(event) for event in result.events[:-1]
|
||||
]
|
||||
assert chunks, f"{AZURE_CHAT_MODEL}: stream held only the [DONE] sentinel"
|
||||
assert all(
|
||||
chunk.object == "chat.completion.chunk" for chunk in chunks
|
||||
), f"{AZURE_CHAT_MODEL}: malformed chunk object types: {result.events[:5]}"
|
||||
assert any(
|
||||
chunk.model for chunk in chunks
|
||||
), f"{AZURE_CHAT_MODEL}: no chunk carried a model name: {result.events[:5]}"
|
||||
|
||||
content = "".join(
|
||||
choice.delta.content or ""
|
||||
for chunk in chunks
|
||||
for choice in chunk.choices
|
||||
if choice.delta is not None
|
||||
)
|
||||
assert content.strip(), (
|
||||
f"{AZURE_CHAT_MODEL}: stream chunks reassembled to an empty "
|
||||
f"completion (#28991): {result.events[:5]}"
|
||||
)
|
||||
|
|
|
|||
|
|
@ -191,6 +191,21 @@ class ChatResponse(BaseModel):
|
|||
service_tier: str | None = None
|
||||
|
||||
|
||||
class ChatStreamDelta(BaseModel):
|
||||
content: str | None = None
|
||||
|
||||
|
||||
class ChatStreamChoice(BaseModel):
|
||||
delta: ChatStreamDelta | None = None
|
||||
|
||||
|
||||
class ChatStreamChunk(BaseModel):
|
||||
id: str | None = None
|
||||
object: str | None = None
|
||||
model: str | None = None
|
||||
choices: list[ChatStreamChoice] = []
|
||||
|
||||
|
||||
class EmbedBody(BaseModel):
|
||||
model: str
|
||||
input: str
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue