mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
test(e2e): tag llm_translation tests with Subject metadata and record harness steps (#44950)
* test(e2e): add enum values, auto-discovering label gates and secret hiding for e2e metadata * test(e2e): tag llm_translation tests with Subject metadata and record harness steps * docs(e2e): name every markerless harness test file that carries no Subject * test(e2e): keep the step discovery comprehensions to one for clause * test(e2e): declare the realtime param tuples Final
This commit is contained in:
parent
736da28ac1
commit
ade17902a0
49 changed files with 2489 additions and 93 deletions
|
|
@ -33,6 +33,8 @@ from anthropic.types import (
|
|||
ToolUseBlockParam,
|
||||
)
|
||||
from e2e_config import provider_edge_base, unique_marker
|
||||
from e2e_metadata import Capability as MetaCapability
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta, step
|
||||
from lifecycle import ResourceManager
|
||||
from llm_translation.sdk_clients import NO_PROXY_CACHE, SdkClients, response_header
|
||||
from models import CredentialCreateBody, LiteLLMParamsBody
|
||||
|
|
@ -65,6 +67,10 @@ Streaming = Literal["stream", "nonstream"]
|
|||
Assertion = Literal["works", "cost_logged"]
|
||||
ToolMode = Literal["none", "forced", "offered"]
|
||||
|
||||
GPT_4O_MINI_BACKEND: Final = "openai/gpt-4o-mini"
|
||||
GPT_5_4_MINI_BACKEND: Final = "openai/gpt-5.4-mini"
|
||||
CLAUDE_HAIKU_BACKEND: Final = "anthropic/claude-haiku-4-5"
|
||||
|
||||
SURFACES: Final[tuple[SurfaceName, ...]] = ("chat_completions", "messages", "responses")
|
||||
AUTH_METHODS: Final[tuple[AuthMethod, ...]] = ("env_ref", "stored_credential")
|
||||
|
||||
|
|
@ -104,12 +110,19 @@ class Deployment:
|
|||
assert key, f"{self.api_key_env} is not set in the test process environment"
|
||||
return key
|
||||
|
||||
def provider(self) -> Provider:
|
||||
match self.route:
|
||||
case "openai":
|
||||
return Provider.OPENAI
|
||||
case "anthropic":
|
||||
return Provider.ANTHROPIC
|
||||
|
||||
|
||||
DEPLOYMENTS: Final[tuple[Deployment, ...]] = (
|
||||
Deployment(
|
||||
route="openai",
|
||||
label="gpt-4o-mini",
|
||||
backend="openai/gpt-4o-mini",
|
||||
backend=GPT_4O_MINI_BACKEND,
|
||||
api_key_env="OPENAI_API_KEY",
|
||||
edge_mount="openai",
|
||||
edge_suffix="/v1",
|
||||
|
|
@ -117,7 +130,7 @@ DEPLOYMENTS: Final[tuple[Deployment, ...]] = (
|
|||
Deployment(
|
||||
route="openai",
|
||||
label="gpt-5.4-mini",
|
||||
backend="openai/gpt-5.4-mini",
|
||||
backend=GPT_5_4_MINI_BACKEND,
|
||||
api_key_env="OPENAI_API_KEY",
|
||||
edge_mount="openai",
|
||||
edge_suffix="/v1",
|
||||
|
|
@ -125,7 +138,7 @@ DEPLOYMENTS: Final[tuple[Deployment, ...]] = (
|
|||
Deployment(
|
||||
route="anthropic",
|
||||
label="claude-haiku-4-5",
|
||||
backend="anthropic/claude-haiku-4-5",
|
||||
backend=CLAUDE_HAIKU_BACKEND,
|
||||
api_key_env="ANTHROPIC_API_KEY",
|
||||
edge_mount="anthropic",
|
||||
edge_suffix="",
|
||||
|
|
@ -146,6 +159,26 @@ class Cell:
|
|||
def registry_id(self, capability: Capability, streaming: Streaming, assertion: Assertion) -> str:
|
||||
return f"llm.{self.surface}.{self.deployment.route}.{capability}.{streaming}.{assertion}"
|
||||
|
||||
def subject(self, capability: Capability, streaming: Streaming, assertion: Assertion) -> Subject:
|
||||
return Subject(
|
||||
domain=Domain.SPEND_BUDGETS if assertion == "cost_logged" else Domain.LLM_TRANSLATION,
|
||||
route=_surface_route(self.surface),
|
||||
providers=(self.deployment.provider(),),
|
||||
models=(self.deployment.backend,),
|
||||
capabilities=() if capability == "basic" else (MetaCapability.FUNCTION_CALLING,),
|
||||
mode=Mode.STREAM if streaming == "stream" else Mode.NONSTREAM,
|
||||
)
|
||||
|
||||
|
||||
def _surface_route(surface: SurfaceName) -> Route:
|
||||
match surface:
|
||||
case "chat_completions":
|
||||
return Route.CHAT_COMPLETIONS
|
||||
case "messages":
|
||||
return Route.MESSAGES
|
||||
case "responses":
|
||||
return Route.RESPONSES
|
||||
|
||||
|
||||
CELLS: Final[tuple[Cell, ...]] = tuple(
|
||||
Cell(surface=surface, deployment=deployment, auth=auth)
|
||||
|
|
@ -158,7 +191,14 @@ CELLS: Final[tuple[Cell, ...]] = tuple(
|
|||
def cells_covering(capability: Capability, streaming: Streaming, assertion: Assertion) -> tuple[ParameterSet, ...]:
|
||||
"""Every cell as a pytest param carrying the registry id its test proves."""
|
||||
return tuple(
|
||||
pytest.param(cell, id=cell.id, marks=pytest.mark.covers(cell.registry_id(capability, streaming, assertion)))
|
||||
pytest.param(
|
||||
cell,
|
||||
id=cell.id,
|
||||
marks=(
|
||||
pytest.mark.covers(cell.registry_id(capability, streaming, assertion)),
|
||||
meta(cell.subject(capability, streaming, assertion)),
|
||||
),
|
||||
)
|
||||
for cell in CELLS
|
||||
)
|
||||
|
||||
|
|
@ -352,9 +392,14 @@ class ChatCompletionsSurface:
|
|||
cost_header=response_header(raw.headers, "x-litellm-response-cost"),
|
||||
)
|
||||
|
||||
@step(
|
||||
'Send a /chat/completions request to {model} with the prompt "{prompt}"'
|
||||
" and forced weather tool use set to {with_tool}"
|
||||
)
|
||||
def reply(self, key: str, model: str, prompt: str, *, with_tool: bool = False) -> Reply:
|
||||
return self._turn(key, model, _chat_history(prompt), "forced" if with_tool else "none")
|
||||
|
||||
@step('Send a streaming /chat/completions request to {model} with the prompt "{prompt}"')
|
||||
def stream(self, key: str, model: str, prompt: str) -> StreamedReply:
|
||||
chunks: Final[tuple[ChatCompletionChunk, ...]] = tuple(
|
||||
self.sdk.openai(key).chat.completions.create(
|
||||
|
|
@ -373,6 +418,7 @@ class ChatCompletionsSurface:
|
|||
event_count=len(chunks),
|
||||
)
|
||||
|
||||
@step("Send the {call.name} tool result back to {model} over /chat/completions")
|
||||
def reply_to_tool_result(self, key: str, model: str, prompt: str, call: ToolCall, result: str) -> Reply:
|
||||
tool_call: Final[ChatCompletionMessageFunctionToolCallParam] = {
|
||||
"id": call.call_id,
|
||||
|
|
@ -421,9 +467,14 @@ class MessagesSurface:
|
|||
cost_header=response_header(raw.headers, "x-litellm-response-cost"),
|
||||
)
|
||||
|
||||
@step(
|
||||
'Send a /v1/messages request to {model} with the prompt "{prompt}"'
|
||||
" and forced weather tool use set to {with_tool}"
|
||||
)
|
||||
def reply(self, key: str, model: str, prompt: str, *, with_tool: bool = False) -> Reply:
|
||||
return self._turn(key, model, ({"role": "user", "content": prompt},), "forced" if with_tool else "none")
|
||||
|
||||
@step('Send a streaming /v1/messages request to {model} with the prompt "{prompt}"')
|
||||
def stream(self, key: str, model: str, prompt: str) -> StreamedReply:
|
||||
events: Final[tuple[RawMessageStreamEvent, ...]] = tuple(
|
||||
self.sdk.anthropic(key).messages.create(
|
||||
|
|
@ -446,6 +497,7 @@ class MessagesSurface:
|
|||
event_count=len(events),
|
||||
)
|
||||
|
||||
@step("Send the {call.name} tool result back to {model} over /v1/messages")
|
||||
def reply_to_tool_result(self, key: str, model: str, prompt: str, call: ToolCall, result: str) -> Reply:
|
||||
tool_use: Final[ToolUseBlockParam] = {
|
||||
"type": "tool_use",
|
||||
|
|
@ -496,9 +548,14 @@ class ResponsesSurface:
|
|||
cost_header=response_header(raw.headers, "x-litellm-response-cost"),
|
||||
)
|
||||
|
||||
@step(
|
||||
'Send a /v1/responses request to {model} with the prompt "{prompt}"'
|
||||
" and forced weather tool use set to {with_tool}"
|
||||
)
|
||||
def reply(self, key: str, model: str, prompt: str, *, with_tool: bool = False) -> Reply:
|
||||
return self._turn(key, model, [{"role": "user", "content": prompt}], "forced" if with_tool else "none")
|
||||
|
||||
@step('Send a streaming /v1/responses request to {model} with the prompt "{prompt}"')
|
||||
def stream(self, key: str, model: str, prompt: str) -> StreamedReply:
|
||||
events: Final[tuple[ResponseStreamEvent, ...]] = tuple(
|
||||
self.sdk.openai(key).responses.create(
|
||||
|
|
@ -519,6 +576,7 @@ class ResponsesSurface:
|
|||
event_count=len(events),
|
||||
)
|
||||
|
||||
@step("Send the {call.name} tool result back to {model} over /v1/responses")
|
||||
def reply_to_tool_result(self, key: str, model: str, prompt: str, call: ToolCall, result: str) -> Reply:
|
||||
function_call: Final[ResponseFunctionToolCallParam] = {
|
||||
"type": "function_call",
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from websockets.exceptions import InvalidStatus
|
|||
from websockets.sync.client import connect
|
||||
|
||||
from e2e_config import ws_base_url
|
||||
from e2e_metadata import step
|
||||
from proxy_client import ProxyClient
|
||||
from e2e_http import FileUploadForm, Headers, NoBody, Result, StreamingResponse
|
||||
from models import ChatMessage
|
||||
|
|
@ -34,7 +35,7 @@ class JsonSchema(BaseModel):
|
|||
|
||||
|
||||
class GeminiHeaders(Headers):
|
||||
x_goog_api_key: str = Field(serialization_alias="x-goog-api-key")
|
||||
x_goog_api_key: str = Field(serialization_alias="x-goog-api-key", repr=False)
|
||||
content_type: str = Field(
|
||||
default="application/json", serialization_alias="Content-Type"
|
||||
)
|
||||
|
|
@ -42,7 +43,7 @@ class GeminiHeaders(Headers):
|
|||
|
||||
|
||||
class AnthropicHeaders(Headers):
|
||||
x_api_key: str = Field(serialization_alias="x-api-key")
|
||||
x_api_key: str = Field(serialization_alias="x-api-key", repr=False)
|
||||
anthropic_version: str = Field(
|
||||
default="2023-06-01", serialization_alias="anthropic-version"
|
||||
)
|
||||
|
|
@ -56,7 +57,7 @@ class VertexHeaders(Headers):
|
|||
# Only the litellm virtual key; the /vertex_ai passthrough mints the Vertex token
|
||||
# from the proxy's own service account (the deployment marked use_in_pass_through),
|
||||
# so no upstream Authorization bearer is sent from the client.
|
||||
x_litellm_api_key: str = Field(serialization_alias="x-litellm-api-key")
|
||||
x_litellm_api_key: str = Field(serialization_alias="x-litellm-api-key", repr=False)
|
||||
content_type: str = Field(
|
||||
default="application/json", serialization_alias="Content-Type"
|
||||
)
|
||||
|
|
@ -246,6 +247,7 @@ class PassthroughClient:
|
|||
|
||||
# ---- Gemini native passthrough (/gemini/v1beta/...) -----------------
|
||||
|
||||
@step("Send a Gemini generateContent request to {model} through /gemini")
|
||||
def gemini_generate(
|
||||
self,
|
||||
key: str,
|
||||
|
|
@ -263,6 +265,7 @@ class PassthroughClient:
|
|||
),
|
||||
)
|
||||
|
||||
@step("Send a Gemini streamGenerateContent request to {model} through /gemini")
|
||||
def gemini_stream(
|
||||
self, key: str, model: str, text: str, *, tags: list[str] | None = None
|
||||
) -> StreamingResponse:
|
||||
|
|
@ -278,6 +281,7 @@ class PassthroughClient:
|
|||
|
||||
# ---- Vertex AI native passthrough (/vertex_ai/v1/projects/...) -------
|
||||
|
||||
@step("Send a Vertex AI generateContent request to {model} in {location} through /vertex_ai")
|
||||
def vertex_generate(
|
||||
self, key: str, project: str, location: str, model: str, text: str
|
||||
) -> StreamingResponse:
|
||||
|
|
@ -295,6 +299,7 @@ class PassthroughClient:
|
|||
|
||||
# ---- Anthropic native passthrough (/anthropic/v1/messages) ----------
|
||||
|
||||
@step("Send a /v1/messages request to {model} through /anthropic with streaming set to {stream}")
|
||||
def anthropic_message(
|
||||
self,
|
||||
key: str,
|
||||
|
|
@ -324,6 +329,7 @@ class PassthroughClient:
|
|||
# Relayed to OpenAI untouched, which is the whole point of the prefix: the
|
||||
# customer opts out of the gateway's managed-file handling here.
|
||||
|
||||
@step("Upload {filename} to /openai_passthrough/v1/files")
|
||||
def openai_passthrough_upload_file(
|
||||
self, key: str, *, content: bytes, filename: str
|
||||
) -> Result[PassthroughFileObject]:
|
||||
|
|
@ -336,6 +342,7 @@ class PassthroughClient:
|
|||
response_type=PassthroughFileObject,
|
||||
)
|
||||
|
||||
@step("Delete the uploaded file through /openai_passthrough/v1/files")
|
||||
def openai_passthrough_delete_file(
|
||||
self, key: str, file_id: str
|
||||
) -> Result[PassthroughFileDeleted]:
|
||||
|
|
@ -346,6 +353,7 @@ class PassthroughClient:
|
|||
response_type=PassthroughFileDeleted,
|
||||
)
|
||||
|
||||
@step("List batches from /openai_passthrough/v1/batches")
|
||||
def openai_passthrough_list_batches(self, key: str) -> Result[PassthroughBatchList]:
|
||||
return self.proxy.transport.get(
|
||||
"/openai_passthrough/v1/batches",
|
||||
|
|
@ -360,6 +368,7 @@ class PassthroughClient:
|
|||
# budgets against this traffic, so a 200 that logs no spend is money the
|
||||
# gateway never sees.
|
||||
|
||||
@step("Send a /v1/responses request to {model} through /openai_passthrough with streaming set to {stream}")
|
||||
def openai_passthrough_responses(
|
||||
self, key: str, model: str, text: str, *, stream: bool = False
|
||||
) -> StreamingResponse:
|
||||
|
|
@ -370,6 +379,7 @@ class PassthroughClient:
|
|||
stream=stream,
|
||||
)
|
||||
|
||||
@step('Send a /v1/embeddings request to {model} through /openai_passthrough for "{text}"')
|
||||
def openai_passthrough_embed(
|
||||
self, key: str, model: str, text: str
|
||||
) -> StreamingResponse:
|
||||
|
|
@ -379,6 +389,7 @@ class PassthroughClient:
|
|||
json=OpenAIEmbeddingBody(model=model, input=text),
|
||||
)
|
||||
|
||||
@step("Send a /v1/chat/completions request to {model} through /openai")
|
||||
def openai_chat(
|
||||
self, key: str, model: str, text: str, *, max_completion_tokens: int = 64
|
||||
) -> StreamingResponse:
|
||||
|
|
@ -397,6 +408,7 @@ class PassthroughClient:
|
|||
# The same prefixes over an upgrade instead of a POST, for the provider APIs
|
||||
# that only speak websocket (realtime, responses.connect).
|
||||
|
||||
@step("Open a websocket to {path} and wait for its first event")
|
||||
def openai_passthrough_websocket(
|
||||
self,
|
||||
key: str,
|
||||
|
|
|
|||
|
|
@ -307,9 +307,11 @@ def as_text(message: str | bytes) -> str:
|
|||
class RealtimeSession:
|
||||
connection: Connection
|
||||
|
||||
def send(self, event: BaseModel) -> None:
|
||||
@step("Send the realtime event {event.type} over the websocket")
|
||||
def send(self, event: SessionUpdate | ConversationItemCreate | ResponseCreate) -> None:
|
||||
self.connection.send(event.model_dump_json(by_alias=True, exclude_none=True))
|
||||
|
||||
@step("Wait for a {stop_type} event on the realtime websocket")
|
||||
def collect_until(
|
||||
self, stop_type: str, *, timeout: float
|
||||
) -> tuple[ReceivedEvent, ...]:
|
||||
|
|
@ -381,6 +383,7 @@ class RealtimeSession:
|
|||
class RealtimeClient:
|
||||
proxy: ProxyClient
|
||||
|
||||
@step("Add a realtime deployment that calls {provider.litellm_params.model}")
|
||||
def provision(self, provider: RealtimeProvider) -> tuple[str, str]:
|
||||
"""Register this provider's realtime deployment through /model/new and return
|
||||
(model_name, model_id). The name is marker-unique so it never collides with a
|
||||
|
|
@ -393,6 +396,7 @@ class RealtimeClient:
|
|||
)
|
||||
return model_name, model_id
|
||||
|
||||
@step("Open a /v1/realtime websocket session to {model}")
|
||||
@contextmanager
|
||||
def connect(
|
||||
self, *, key: str, model: str, timeout: float = 15.0
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ from __future__ import annotations
|
|||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from realtime_client import (
|
||||
|
|
@ -42,6 +43,15 @@ class TestNovaSonicRealtime:
|
|||
"llm.realtime.bedrock_converse.basic.stream.works",
|
||||
exercised_on=["realtime"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(NOVA_SONIC,),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
)
|
||||
def test_nova_sonic_response_create_completes(
|
||||
self, client: RealtimeClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -12,7 +12,10 @@ hard failure, not a skip; once configured, a protocol failure is likewise a hard
|
|||
failure. See REALTIME_COVERAGE_MATRIX.md.
|
||||
"""
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from pydantic import BaseModel
|
||||
|
|
@ -42,7 +45,42 @@ from websockets.exceptions import ConnectionClosedError
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
PROVIDER_PARAMS = [pytest.param(p, id=p.id) for p in PROVIDERS]
|
||||
AZURE_REALTIME_MODEL: Final = "azure/gpt-realtime"
|
||||
|
||||
TEXT_PARAMS: Final = tuple(
|
||||
pytest.param(
|
||||
p,
|
||||
id=p.id,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider(p.id),),
|
||||
models=(p.litellm_params.model,),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
),
|
||||
)
|
||||
for p in PROVIDERS
|
||||
)
|
||||
|
||||
TOOL_PARAMS: Final = tuple(
|
||||
pytest.param(
|
||||
p,
|
||||
id=p.id,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider(p.id),),
|
||||
models=(p.litellm_params.model,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
),
|
||||
)
|
||||
for p in PROVIDERS
|
||||
)
|
||||
|
||||
WEATHER_TOOL = FunctionTool(
|
||||
name="get_weather",
|
||||
|
|
@ -62,7 +100,7 @@ class WeatherResult(BaseModel):
|
|||
temperature_f: int
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", PROVIDER_PARAMS)
|
||||
@pytest.mark.parametrize("provider", TEXT_PARAMS)
|
||||
def test_text_conversation(
|
||||
client: RealtimeClient,
|
||||
scoped_key: str,
|
||||
|
|
@ -99,7 +137,7 @@ def test_text_conversation(
|
|||
assert done.response.usage is not None, "response.done missing normalized usage"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", PROVIDER_PARAMS)
|
||||
@pytest.mark.parametrize("provider", TOOL_PARAMS)
|
||||
def test_tool_call_round_trip(
|
||||
client: RealtimeClient,
|
||||
scoped_key: str,
|
||||
|
|
@ -158,7 +196,7 @@ _REFUSED_UPSTREAMS = (
|
|||
"azure-bad-key",
|
||||
"azure-realtime-refused",
|
||||
LiteLLMParamsBody(
|
||||
model="azure/gpt-realtime",
|
||||
model=AZURE_REALTIME_MODEL,
|
||||
api_key="invalid-e2e-key",
|
||||
api_version="2025-08-28",
|
||||
realtime_protocol="GA",
|
||||
|
|
@ -168,6 +206,15 @@ _REFUSED_UPSTREAMS = (
|
|||
|
||||
|
||||
@pytest.mark.parametrize("provider", _REFUSED_UPSTREAMS, ids=[p.id for p in _REFUSED_UPSTREAMS])
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_REALTIME_MODEL,),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
)
|
||||
def test_upstream_handshake_refusal_is_an_error_event_and_policy_close(
|
||||
client: RealtimeClient,
|
||||
resources: ResourceManager,
|
||||
|
|
|
|||
|
|
@ -24,10 +24,12 @@ Three test scenarios per provider:
|
|||
import asyncio
|
||||
import wave
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import ws_base_url
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from realtime_client import (
|
||||
PROVIDERS,
|
||||
RealtimeProvider,
|
||||
|
|
@ -73,7 +75,59 @@ from pipecat.services.openai.realtime.llm import OpenAIRealtimeLLMService # noq
|
|||
|
||||
from pipecat_service import LiteLLMRealtimeLLMService # noqa: E402
|
||||
|
||||
PROVIDER_PARAMS = [pytest.param(p, id=p.id) for p in PROVIDERS]
|
||||
TOOL_PARAMS: Final = tuple(
|
||||
pytest.param(
|
||||
p,
|
||||
id=p.id,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider(p.id),),
|
||||
models=(p.litellm_params.model,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
),
|
||||
)
|
||||
for p in PROVIDERS
|
||||
)
|
||||
|
||||
AUDIO_OUTPUT_PARAMS: Final = tuple(
|
||||
pytest.param(
|
||||
p,
|
||||
id=p.id,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider(p.id),),
|
||||
models=(p.litellm_params.model,),
|
||||
capabilities=(Capability.AUDIO_OUTPUT,),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
),
|
||||
)
|
||||
for p in PROVIDERS
|
||||
)
|
||||
|
||||
AUDIO_INPUT_PARAMS: Final = tuple(
|
||||
pytest.param(
|
||||
p,
|
||||
id=p.id,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider(p.id),),
|
||||
models=(p.litellm_params.model,),
|
||||
capabilities=(Capability.AUDIO_INPUT, Capability.AUDIO_OUTPUT),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
),
|
||||
)
|
||||
for p in PROVIDERS
|
||||
)
|
||||
|
||||
# PCM16 24 kHz mono WAV of "What is the weather in Paris?" (generated via macOS
|
||||
# `say` and resampled with audioop). Used by the server-VAD audio-input test.
|
||||
|
|
@ -193,7 +247,7 @@ async def _run_pipeline(
|
|||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", PROVIDER_PARAMS)
|
||||
@pytest.mark.parametrize("provider", TOOL_PARAMS)
|
||||
def test_pipecat_server_vad(
|
||||
scoped_key: str,
|
||||
realtime_models: dict[str, str],
|
||||
|
|
@ -208,7 +262,7 @@ def test_pipecat_server_vad(
|
|||
assert got_text, "no assistant text frames produced"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", PROVIDER_PARAMS)
|
||||
@pytest.mark.parametrize("provider", AUDIO_OUTPUT_PARAMS)
|
||||
def test_pipecat_audio_output(
|
||||
scoped_key: str,
|
||||
realtime_models: dict[str, str],
|
||||
|
|
@ -332,7 +386,7 @@ async def _run_audio_input_pipeline(
|
|||
return bool(capture.texts), capture.audio_bytes
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", PROVIDER_PARAMS)
|
||||
@pytest.mark.parametrize("provider", AUDIO_INPUT_PARAMS)
|
||||
def test_pipecat_server_vad_audio_input(
|
||||
scoped_key: str,
|
||||
realtime_models: dict[str, str],
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ import asyncio
|
|||
import pytest
|
||||
|
||||
from e2e_config import ws_base_url
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from realtime_client import (
|
||||
PROVIDERS,
|
||||
RealtimeProvider,
|
||||
|
|
@ -64,7 +65,21 @@ from pipecat_service import LiteLLMRealtimeLLMService # noqa: E402
|
|||
# pipecat-ai/pipecat#2544); raw-ws tool_call_round_trip[vertex_ai] is the
|
||||
# source of truth for that provider. Keep openai/azure/gemini here.
|
||||
PROVIDER_PARAMS = [
|
||||
pytest.param(p, id=p.id) for p in PROVIDERS if p.id != "vertex_ai"
|
||||
pytest.param(
|
||||
p,
|
||||
id=p.id,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider(p.id),),
|
||||
models=(p.litellm_params.model,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
),
|
||||
)
|
||||
for p in PROVIDERS if p.id != "vertex_ai"
|
||||
]
|
||||
|
||||
WEATHER_TOOL = ToolsSchema(
|
||||
|
|
|
|||
|
|
@ -10,8 +10,11 @@ the SDK refuses to send a request missing its required fields.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from e2e_http import assert_client_error
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
|
|
@ -21,6 +24,9 @@ from sdk_clients import SdkClients, response_header
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
OPENAI_TTS_MODEL: Final = "openai/gpt-4o-mini-tts"
|
||||
AWS_POLLY_MODEL: Final = "aws_polly/generative"
|
||||
|
||||
|
||||
class _OptionalSpeechBody(BaseModel):
|
||||
model: str | None = None
|
||||
|
|
@ -32,7 +38,7 @@ def _register_tts(proxy: ProxyClient, resources: ResourceManager) -> tuple[str,
|
|||
model = f"e2e-speech-{unique_marker()}"
|
||||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(model="openai/gpt-4o-mini-tts", api_key="os.environ/OPENAI_API_KEY"),
|
||||
LiteLLMParamsBody(model=OPENAI_TTS_MODEL, api_key="os.environ/OPENAI_API_KEY"),
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
return model, resources.key()
|
||||
|
|
@ -40,6 +46,15 @@ def _register_tts(proxy: ProxyClient, resources: ResourceManager) -> tuple[str,
|
|||
|
||||
class TestAudioSpeech:
|
||||
@pytest.mark.covers("llm.audio_speech.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_TTS_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_audio_speech_returns_audio(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -56,6 +71,15 @@ class TestAudioSpeech:
|
|||
assert response.content, "/audio/speech returned an empty body"
|
||||
|
||||
@pytest.mark.covers("llm.audio_speech.openai.basic.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_TTS_MODEL,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_audio_speech_streams_audio_chunks(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -90,6 +114,15 @@ class TestAudioSpeech:
|
|||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /v1/audio/speech 500s on missing input instead of 400")
|
||||
@pytest.mark.covers("llm.audio_speech.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_TTS_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_missing_input_returns_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -103,6 +136,12 @@ class TestAudioSpeech:
|
|||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /v1/audio/speech 500s on missing model instead of 400")
|
||||
@pytest.mark.covers("llm.audio_speech.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
)
|
||||
)
|
||||
def test_missing_model_returns_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -116,6 +155,15 @@ class TestAudioSpeech:
|
|||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /v1/audio/speech 500s on invalid voice instead of surfacing the provider 4xx")
|
||||
@pytest.mark.covers("llm.audio_speech.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_TTS_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invalid_voice_returns_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -129,6 +177,15 @@ class TestAudioSpeech:
|
|||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /v1/audio/speech 500s on empty input instead of surfacing the provider 4xx")
|
||||
@pytest.mark.covers("llm.audio_speech.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_TTS_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_empty_input_returns_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -145,6 +202,15 @@ MP3_PREFIXES = (b"ID3", b"\xff\xfb", b"\xff\xf3", b"\xff\xf2")
|
|||
|
||||
|
||||
class TestAwsPollySpeech:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.AWS_POLLY,),
|
||||
models=(AWS_POLLY_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_polly_generative_voice_returns_mp3(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -152,7 +218,7 @@ class TestAwsPollySpeech:
|
|||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="aws_polly/generative",
|
||||
model=AWS_POLLY_MODEL,
|
||||
aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID",
|
||||
aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY",
|
||||
aws_region_name="os.environ/AWS_REGION",
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from typing import Final
|
|||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from e2e_http import UnknownApiError, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
|
|
@ -30,6 +31,8 @@ WEATHER_WAV = (
|
|||
Path(__file__).resolve().parent / "realtime" / "fixtures" / "weather_question_24k.wav"
|
||||
)
|
||||
|
||||
OPENAI_TRANSCRIBE_MODEL: Final = "openai/gpt-4o-mini-transcribe"
|
||||
OPENAI_WHISPER_MODEL: Final = "openai/whisper-1"
|
||||
MISSING_MODEL_PHRASES: Final = ("model=none", "invalid model", "model is required")
|
||||
|
||||
|
||||
|
|
@ -47,7 +50,7 @@ def _register(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, str]
|
|||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/gpt-4o-mini-transcribe", api_key="os.environ/OPENAI_API_KEY"
|
||||
model=OPENAI_TRANSCRIBE_MODEL, api_key="os.environ/OPENAI_API_KEY"
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
|
|
@ -56,6 +59,15 @@ def _register(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, str]
|
|||
|
||||
class TestAudioTranscriptions:
|
||||
@pytest.mark.covers("llm.audio_transcriptions.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_TRANSCRIBE_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_audio_transcriptions_returns_text(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -72,6 +84,15 @@ class TestAudioTranscriptions:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.audio_transcriptions.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_TRANSCRIBE_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_missing_file_returns_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -98,6 +119,12 @@ class TestAudioTranscriptions:
|
|||
pytest.fail(f"empty audio expected a file-specific 400, got {other!r}")
|
||||
|
||||
@pytest.mark.covers("llm.audio_transcriptions.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
)
|
||||
)
|
||||
def test_missing_model_returns_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -143,7 +170,7 @@ class TestWhisperTranscriptionFormats:
|
|||
self, proxy: ProxyClient, resources: ResourceManager, form: _WhisperForm, response_type: type[R]
|
||||
) -> R:
|
||||
model_id = proxy.create_model(
|
||||
form.model, LiteLLMParamsBody(model="openai/whisper-1", api_key="os.environ/OPENAI_API_KEY")
|
||||
form.model, LiteLLMParamsBody(model=OPENAI_WHISPER_MODEL, api_key="os.environ/OPENAI_API_KEY")
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
return unwrap(
|
||||
|
|
@ -158,12 +185,30 @@ class TestWhisperTranscriptionFormats:
|
|||
)
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_WHISPER_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vtt_format_returns_webvtt_transcript(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
form = _WhisperForm(model=f"e2e-whisper-vtt-{unique_marker()}", response_format="vtt")
|
||||
transcript = self._upload(proxy, resources, form, _TranscriptionResult)
|
||||
assert transcript.text.lstrip().startswith("WEBVTT"), f"vtt transcript is not WebVTT: {transcript.text[:200]!r}"
|
||||
assert "weather" in transcript.text.lower(), f"vtt transcript lost the spoken words: {transcript.text!r}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.AUDIO,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_WHISPER_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_verbose_json_returns_word_timestamps(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
form = _WhisperForm(
|
||||
model=f"e2e-whisper-verbose-{unique_marker()}",
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ from __future__ import annotations
|
|||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from e2e_http import (
|
||||
assert_client_error,
|
||||
require_successful_call,
|
||||
|
|
@ -100,6 +101,15 @@ def _default_invoke() -> InvokeBody:
|
|||
|
||||
class TestBedrockNative:
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_converse.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_converse_returns_assistant(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -113,6 +123,15 @@ class TestBedrockNative:
|
|||
assert any(part.text.strip() for part in response.output.message.content)
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_converse.basic.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_converse_stream_returns_chunks(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -126,6 +145,15 @@ class TestBedrockNative:
|
|||
assert result.chunks > 0, "converse-stream returned no events"
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_invoke.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invoke_returns_message(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -138,6 +166,15 @@ class TestBedrockNative:
|
|||
assert any(part.text.strip() for part in response.content)
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_invoke.basic.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_invoke_stream_returns_chunks(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -151,6 +188,15 @@ class TestBedrockNative:
|
|||
assert result.chunks > 0, "invoke stream returned no events"
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_converse.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_converse_missing_messages_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -161,6 +207,15 @@ class TestBedrockNative:
|
|||
assert_client_error(result, "converse missing messages")
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_converse.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_converse_empty_messages_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -171,6 +226,14 @@ class TestBedrockNative:
|
|||
assert_client_error(result, "converse empty messages")
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_converse.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_converse_invalid_model_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
_, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -183,6 +246,15 @@ class TestBedrockNative:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_invoke.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invoke_missing_messages_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -193,6 +265,15 @@ class TestBedrockNative:
|
|||
assert_client_error(result, "invoke missing messages")
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_invoke.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invoke_missing_max_tokens_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -206,6 +287,15 @@ class TestBedrockNative:
|
|||
assert_client_error(result, "invoke missing max_tokens")
|
||||
|
||||
@pytest.mark.covers("llm.bedrock_native.bedrock_invoke.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invoke_invalid_temperature_returns_client_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ import pytest
|
|||
from pydantic import BaseModel
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from e2e_http import StreamingResponse, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody
|
||||
|
|
@ -98,6 +99,15 @@ class TestBedrockResponseHeaders:
|
|||
"llm.chat_completions.bedrock_converse.response_headers.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(CONVERSE_REGIONAL_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_request_id_header_surfaces(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -117,6 +127,15 @@ class TestBedrockResponseHeaders:
|
|||
"llm.chat_completions.bedrock_converse.response_headers.stream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(CONVERSE_REGIONAL_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_request_id_header_surfaces_on_stream(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -158,6 +177,15 @@ class TestBedrockBatchDeploymentServesChat:
|
|||
"llm.chat_completions.bedrock_converse.batch_deployment.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(CONVERSE_REGIONAL_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_batch_s3_keys_do_not_break_chat(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -179,6 +207,15 @@ class TestBedrockBatchDeploymentServesChat:
|
|||
|
||||
class TestBedrockInvokeRegionalModelIds:
|
||||
@pytest.mark.covers("llm.chat_completions.bedrock_invoke.basic.nonstream.works", exercised_on=[])
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(INVOKE_REGIONAL_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invoke_regional_id_completes(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -190,6 +227,15 @@ class TestBedrockInvokeRegionalModelIds:
|
|||
_assert_completion(response)
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.bedrock_invoke.basic.stream.works", exercised_on=[])
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(INVOKE_REGIONAL_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_invoke_regional_id_streams(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -205,6 +251,15 @@ class TestBedrockInvokeRegionalModelIds:
|
|||
|
||||
class TestBedrockOpenAIFamilyDefaultRoute:
|
||||
@pytest.mark.covers("llm.chat_completions.bedrock_converse.basic.nonstream.works", exercised_on=[])
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(OPENAI_FAMILY_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_family_model_id_completes_with_max_tokens(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -36,6 +36,7 @@ from __future__ import annotations
|
|||
import pytest
|
||||
from anthropic.types import WebSearchTool20250305Param
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -62,6 +63,16 @@ class TestBedrockWebSearchServerTool:
|
|||
"ephemeral stack ships the config in this module's docstring."
|
||||
)
|
||||
@pytest.mark.covers("llm.messages.bedrock_invoke.web_search_server_tool.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_INVOKE_BACKEND,),
|
||||
capabilities=(Capability.WEB_SEARCH,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_web_search_server_tool_is_served(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -43,6 +43,7 @@ import pytest
|
|||
from pydantic import BaseModel
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from e2e_http import Result, UnknownApiError, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import CacheControl, ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody, RichMessage, TextBlock, Usage
|
||||
|
|
@ -226,6 +227,16 @@ class TestCacheControl:
|
|||
"llm.chat_completions.bedrock_converse.prompt_cache_5m.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_MODEL,),
|
||||
capabilities=(Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_prompt_caching_reads_cache(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -242,6 +253,16 @@ class TestCacheControl:
|
|||
"llm.chat_completions.vertex.prompt_cache_5m.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_MODEL,),
|
||||
capabilities=(Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_prompt_caching_reads_cache(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -266,6 +287,16 @@ class TestCacheControl:
|
|||
"llm.chat_completions.anthropic.prompt_cache_5m.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_MODEL,),
|
||||
capabilities=(Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_prompt_caching_reads_cache(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -282,6 +313,16 @@ class TestCacheControl:
|
|||
"llm.chat_completions.openai.prompt_cache_5m.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MODEL,),
|
||||
capabilities=(Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_prompt_caching_reads_cache(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ from typing import Final, Literal, TypeAlias
|
|||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import Result, unwrap
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import (
|
||||
CacheControl,
|
||||
|
|
@ -162,7 +163,36 @@ def _assert_normal_completion(response: ChatResponse, model_name: str) -> None:
|
|||
|
||||
@pytest.mark.parametrize(
|
||||
"backend",
|
||||
(pytest.param("azure_foundry", id="azure-foundry"), pytest.param("vertex", id="vertex")),
|
||||
(
|
||||
pytest.param(
|
||||
"azure_foundry",
|
||||
id="azure-foundry",
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(AZURE_MODEL,),
|
||||
capabilities=(Capability.FUNCTION_CALLING, Capability.PROMPT_CACHING),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
),
|
||||
),
|
||||
pytest.param(
|
||||
"vertex",
|
||||
id="vertex",
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_MODEL,),
|
||||
capabilities=(Capability.FUNCTION_CALLING, Capability.PROMPT_CACHING),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
@pytest.mark.provider_live
|
||||
@pytest.mark.covers("llm.chat_completions.azure_foundry.basic.nonstream.works")
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ from __future__ import annotations
|
|||
import pytest
|
||||
from e2e_config import provider_edge_base, unique_marker
|
||||
from e2e_http import StreamingResponse, assert_client_error, require_successful_call, unwrap
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -62,6 +63,15 @@ def _chat_status(proxy: ProxyClient, key: str, body: BaseModel) -> StreamingResp
|
|||
|
||||
class TestChatCompletionsContract:
|
||||
@pytest.mark.covers("llm.chat_completions.openai.multi_turn.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_multi_turn_history_is_honored(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_chat_model(proxy, resources)
|
||||
turn1 = unwrap(
|
||||
|
|
@ -106,6 +116,15 @@ class TestChatCompletionsContract:
|
|||
assert "84" in second, f"turn2 must answer 84 from history, got: {second!r}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_success_response_matches_chat_completion_contract(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -131,6 +150,12 @@ class TestChatCompletionsContract:
|
|||
assert (message.content or "").strip(), f"content must be non-empty: {result.body[:300]}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
)
|
||||
)
|
||||
def test_missing_model_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
_, key = _register_chat_model(proxy, resources)
|
||||
result = _chat_status(
|
||||
|
|
@ -145,12 +170,30 @@ class TestChatCompletionsContract:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_missing_messages_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_chat_model(proxy, resources)
|
||||
result = _chat_status(proxy, key, ChatMissingMessagesBody(model=model))
|
||||
assert_client_error(result, "missing messages")
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_empty_messages_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_chat_model(proxy, resources)
|
||||
result = _chat_status(
|
||||
|
|
@ -161,6 +204,15 @@ class TestChatCompletionsContract:
|
|||
assert_client_error(result, "empty messages")
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invalid_role_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_chat_model(proxy, resources)
|
||||
result = _chat_status(
|
||||
|
|
@ -175,6 +227,15 @@ class TestChatCompletionsContract:
|
|||
assert_client_error(result, "invalid role")
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invalid_temperatures_return_client_errors(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_chat_model(proxy, resources)
|
||||
for temperature in (-0.1, 2.1, 3.0, 100.0):
|
||||
|
|
@ -191,6 +252,15 @@ class TestChatCompletionsContract:
|
|||
assert_client_error(result, f"temperature={temperature}")
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invalid_max_completion_tokens_return_client_errors(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -208,6 +278,15 @@ class TestChatCompletionsContract:
|
|||
assert_client_error(result, f"max_completion_tokens={max_completion_tokens}")
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_temperature_boundaries_succeed(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_chat_model(proxy, resources)
|
||||
for temperature in (0.0, 2.0):
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ from pydantic import BaseModel
|
|||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import StreamingResponse, unwrap
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import (
|
||||
ChatBody,
|
||||
|
|
@ -53,16 +54,40 @@ OPENAI_BACKEND = "openai/gpt-5.6"
|
|||
ANTHROPIC_BACKEND = "anthropic/claude-haiku-4-5-20251001"
|
||||
BEDROCK_CONVERSE_BACKEND = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"
|
||||
BEDROCK_NOVA_BACKEND: Final = "bedrock/us.amazon.nova-2-lite-v1:0"
|
||||
VERTEX_MISTRAL_BACKEND: Final = "vertex_ai/mistral-small-2503"
|
||||
VERTEX_GPT_OSS_BACKEND: Final = "vertex_ai/openai/gpt-oss-120b-maas"
|
||||
VERTEX_PARTNER_BACKENDS: Final = (
|
||||
pytest.param(
|
||||
"vertex_ai/mistral-small-2503",
|
||||
marks=pytest.mark.skip(
|
||||
reason="the e2e Vertex project has no access to mistral-small-2503 (404 publisher model not found)"
|
||||
VERTEX_MISTRAL_BACKEND,
|
||||
marks=(
|
||||
pytest.mark.skip(
|
||||
reason="the e2e Vertex project has no access to mistral-small-2503 (404 publisher model not found)"
|
||||
),
|
||||
meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_MISTRAL_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
),
|
||||
),
|
||||
),
|
||||
pytest.param(
|
||||
"vertex_ai/openai/gpt-oss-120b-maas",
|
||||
marks=pytest.mark.skip(reason="never served by the e2e Vertex project (60s read timeout, no headers)"),
|
||||
VERTEX_GPT_OSS_BACKEND,
|
||||
marks=(
|
||||
pytest.mark.skip(reason="never served by the e2e Vertex project (60s read timeout, no headers)"),
|
||||
meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_GPT_OSS_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
PDF_DOCUMENT_URL: Final = (
|
||||
|
|
@ -216,19 +241,58 @@ _PERSON_SCHEMA: dict[str, object] = {
|
|||
},
|
||||
}
|
||||
|
||||
CHAT_MODELS: tuple[tuple[str, str], ...] = (
|
||||
("gpt-5.5", "openai"),
|
||||
("claude-haiku-4-5", "anthropic"),
|
||||
("gemini-2.5-flash", "gemini"),
|
||||
OPENAI_CHAT_MODEL: Final = "gpt-5.5"
|
||||
ANTHROPIC_CHAT_MODEL: Final = "claude-haiku-4-5"
|
||||
GEMINI_FLASH_MODEL: Final = "gemini-2.5-flash"
|
||||
|
||||
CHAT_MODELS: Final = (
|
||||
pytest.param(
|
||||
OPENAI_CHAT_MODEL,
|
||||
"openai",
|
||||
id=f"{OPENAI_CHAT_MODEL}-openai",
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_CHAT_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
),
|
||||
),
|
||||
pytest.param(
|
||||
ANTHROPIC_CHAT_MODEL,
|
||||
"anthropic",
|
||||
id=f"{ANTHROPIC_CHAT_MODEL}-anthropic",
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_CHAT_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
),
|
||||
),
|
||||
pytest.param(
|
||||
GEMINI_FLASH_MODEL,
|
||||
"gemini",
|
||||
id=f"{GEMINI_FLASH_MODEL}-gemini",
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_FLASH_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class TestChatCompletionsRegression:
|
||||
@pytest.mark.parametrize(
|
||||
("model", "route"),
|
||||
CHAT_MODELS,
|
||||
ids=[f"{model}-{route}" for model, route in CHAT_MODELS],
|
||||
)
|
||||
@pytest.mark.parametrize(("model", "route"), CHAT_MODELS)
|
||||
@pytest.mark.covers(
|
||||
"llm.chat_completions.openai.basic.nonstream.works",
|
||||
"llm.chat_completions.anthropic.basic.nonstream.works",
|
||||
|
|
@ -272,6 +336,15 @@ class TestCohereChat:
|
|||
"llm.chat_completions.cohere.basic.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.COHERE,),
|
||||
models=(COHERE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_cohere_chat_returns_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -316,6 +389,15 @@ class TestGeminiChatCompletions:
|
|||
"llm.chat_completions.gemini.basic.nonstream.cost_logged",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_gemini_chat_returns_content_and_logs_cost(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -377,6 +459,15 @@ class TestVertexChatCompletions:
|
|||
"llm.chat_completions.vertex.basic.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_chat_returns_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -406,6 +497,16 @@ class TestVertexChatCompletions:
|
|||
"llm.chat_completions.vertex.tool_use.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_chat_returns_tool_call(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -435,6 +536,16 @@ class TestVertexChatCompletions:
|
|||
"llm.chat_completions.vertex.vision.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_BACKEND,),
|
||||
capabilities=(Capability.VISION,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_chat_vision_describes_image(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -452,6 +563,15 @@ class TestVertexChatCompletions:
|
|||
"llm.chat_completions.vertex.basic.stream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_chat_streams_real_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -494,6 +614,15 @@ class TestAzureOpenAIChatCompletions:
|
|||
"llm.chat_completions.azure_openai.basic.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_azure_openai_chat_returns_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -523,6 +652,16 @@ class TestAzureOpenAIChatCompletions:
|
|||
"llm.chat_completions.azure_openai.tool_use.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_OPENAI_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_azure_openai_chat_returns_tool_call(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -554,6 +693,15 @@ class TestAzureFoundryChatCompletions:
|
|||
"llm.chat_completions.azure_foundry.basic.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(AZURE_FOUNDRY_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_azure_foundry_chat_returns_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -596,6 +744,14 @@ class TestHostedVllmChat:
|
|||
"llm.chat_completions.hosted_vllm.basic.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.HOSTED_VLLM,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_hosted_vllm_chat_returns_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -651,6 +807,15 @@ class TestOpenAIChatCompletions:
|
|||
"llm.chat_completions.openai.basic.stream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_chat_streams_real_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -678,6 +843,15 @@ class TestOpenAIChatCompletions:
|
|||
"llm.chat_completions.openai.basic.nonstream.cost_logged",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_chat_logs_cost(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -711,6 +885,16 @@ class TestOpenAIChatCompletions:
|
|||
"llm.chat_completions.openai.tool_use.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_chat_returns_tool_call(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -742,6 +926,16 @@ class TestOpenAIChatCompletions:
|
|||
"llm.chat_completions.openai.structured_output.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
capabilities=(Capability.RESPONSE_SCHEMA,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_chat_structured_output_conforms_to_schema(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -775,6 +969,16 @@ class TestOpenAIChatCompletions:
|
|||
"llm.chat_completions.openai.thinking.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_chat_reasoning_reports_reasoning_tokens(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -825,6 +1029,16 @@ class TestOpenAIChatCompletions:
|
|||
"llm.chat_completions.openai.vision.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_VISION_BACKEND,),
|
||||
capabilities=(Capability.VISION,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_chat_vision_describes_image(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -842,6 +1056,16 @@ class TestOpenAIChatCompletions:
|
|||
"llm.chat_completions.openai.tool_use.stream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_chat_streams_tool_call(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -889,6 +1113,15 @@ class TestBedrockConverseChatCompletions:
|
|||
"llm.chat_completions.bedrock_converse.basic.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_converse_chat_returns_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -913,6 +1146,15 @@ class TestBedrockConverseChatCompletions:
|
|||
"llm.chat_completions.bedrock_converse.basic.stream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_converse_chat_streams_real_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -936,6 +1178,16 @@ class TestBedrockConverseChatCompletions:
|
|||
"llm.chat_completions.bedrock_converse.tool_use.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_converse_chat_returns_tool_call(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -962,6 +1214,16 @@ class TestBedrockConverseChatCompletions:
|
|||
"llm.chat_completions.bedrock_converse.thinking.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_converse_chat_returns_reasoning(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -992,6 +1254,16 @@ class TestBedrockConverseChatCompletions:
|
|||
"llm.chat_completions.bedrock_converse.vision.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
capabilities=(Capability.VISION,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_converse_chat_vision_describes_image(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -1001,6 +1273,16 @@ class TestBedrockConverseChatCompletions:
|
|||
response = unwrap(client.proxy.chat(key, ChatBody(model=model, messages=_vision_messages(), max_tokens=32)))
|
||||
_assert_describes_cat(response)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_NOVA_BACKEND,),
|
||||
capabilities=(Capability.PDF_INPUT,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_converse_reads_a_pdf_sent_by_url(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -1107,6 +1389,16 @@ class TestAnthropicChatCompletions:
|
|||
"llm.chat_completions.anthropic.structured_output.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.RESPONSE_SCHEMA,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_chat_structured_output_conforms_to_schema(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -1140,6 +1432,16 @@ class TestAnthropicChatCompletions:
|
|||
"llm.chat_completions.anthropic.thinking.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_chat_returns_thinking_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -1178,6 +1480,16 @@ class TestAnthropicChatCompletions:
|
|||
"llm.chat_completions.anthropic.vision.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.VISION,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_chat_vision_describes_image(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -1191,6 +1503,15 @@ class TestAnthropicChatCompletions:
|
|||
"llm.chat_completions.anthropic.basic.stream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_chat_streams_real_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -1214,6 +1535,16 @@ class TestAnthropicChatCompletions:
|
|||
"llm.chat_completions.anthropic.tool_use.nonstream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_chat_returns_tool_call(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -1240,6 +1571,16 @@ class TestAnthropicChatCompletions:
|
|||
"llm.chat_completions.anthropic.tool_use.stream.works",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_chat_streams_tool_call(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -37,6 +37,7 @@ from pydantic import BaseModel
|
|||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import Result, unwrap
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import CacheControl, ChatResponse, LiteLLMParamsBody, RichMessage, TextBlock, Usage
|
||||
from passthrough_client import PassthroughClient
|
||||
|
|
@ -281,6 +282,16 @@ class TestAnthropicChatMidConversationSystem:
|
|||
"llm.chat_completions.anthropic.mid_conversation_system.nonstream.cache_hit",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(FLAGGED_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -290,6 +301,16 @@ class TestAnthropicChatMidConversationSystem:
|
|||
"llm.chat_completions.anthropic.mid_conversation_system.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(UNFLAGGED_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_unflagged_model_converts_system_reminder_and_succeeds(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -305,6 +326,16 @@ class TestBedrockInvokeChatMidConversationSystem:
|
|||
"llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.cache_hit",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(FLAGGED_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -314,6 +345,16 @@ class TestBedrockInvokeChatMidConversationSystem:
|
|||
"llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(UNFLAGGED_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_unflagged_model_converts_system_reminder_and_succeeds(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ from typing import Final
|
|||
import pytest
|
||||
from e2e_config import provider_edge_base, unique_marker
|
||||
from e2e_http import require_successful_call
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatStreamOptions, LiteLLMParamsBody, Usage
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -12,6 +13,8 @@ from pydantic import BaseModel
|
|||
|
||||
pytestmark = [pytest.mark.e2e, pytest.mark.replayable]
|
||||
|
||||
OPENAI_BACKEND: Final = "openai/gpt-5.6"
|
||||
|
||||
|
||||
class _Delta(BaseModel):
|
||||
content: str | None = None
|
||||
|
|
@ -30,13 +33,22 @@ class _Chunk(BaseModel):
|
|||
|
||||
class TestChatStreamContract:
|
||||
@pytest.mark.covers("llm.chat_completions.openai.basic.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_chat_stream_is_sse_and_ends_with_done(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model: Final = f"e2e-chat-stream-{unique_marker()}"
|
||||
base: Final = provider_edge_base("openai")
|
||||
model_id: Final = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/gpt-5.6",
|
||||
model=OPENAI_BACKEND,
|
||||
api_key="os.environ/OPENAI_API_KEY",
|
||||
api_base=f"{base}/v1" if base else None,
|
||||
),
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ from typing import Final
|
|||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import (
|
||||
ChatAssistantTurn,
|
||||
|
|
@ -138,22 +139,72 @@ def _assert_tool_results_reach_the_model(
|
|||
|
||||
|
||||
class TestChatToolResultRoundTrip:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_gemini(self, client: PassthroughClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(client, resources, _api_key_params(GEMINI_BACKEND, "GEMINI_API_KEY"))
|
||||
_assert_tool_results_reach_the_model(client, key, model, thinking=None, tool_choice="required")
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.MISTRAL,),
|
||||
models=(MISTRAL_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_mistral(self, client: PassthroughClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(client, resources, _api_key_params(MISTRAL_BACKEND, "MISTRAL_API_KEY"))
|
||||
_assert_tool_results_reach_the_model(client, key, model, thinking=None, tool_choice="required")
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_converse(self, client: PassthroughClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(client, resources, _bedrock_params(BEDROCK_CONVERSE_BACKEND))
|
||||
_assert_tool_results_reach_the_model(client, key, model, thinking=None, tool_choice="required")
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING, Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_with_extended_thinking(self, client: PassthroughClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(client, resources, _api_key_params(ANTHROPIC_BACKEND, "ANTHROPIC_API_KEY"))
|
||||
_assert_tool_results_reach_the_model(client, key, model, thinking=THINKING, tool_choice=None)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_LEGACY_THINKING_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING, Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_converse_with_extended_thinking(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -10,8 +10,11 @@ the completion fails here.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -19,9 +22,20 @@ from sdk_clients import NO_PROXY_CACHE, SdkClients
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
OPENAI_COMPLETIONS_BACKEND: Final = "openai/gpt-5.4-nano"
|
||||
|
||||
|
||||
class TestCompletionsEndpoint:
|
||||
@pytest.mark.covers("llm.completions.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_COMPLETIONS_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_text_completion_returns_text(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -29,7 +43,7 @@ class TestCompletionsEndpoint:
|
|||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/gpt-5.4-nano",
|
||||
model=OPENAI_COMPLETIONS_BACKEND,
|
||||
api_key="os.environ/OPENAI_API_KEY",
|
||||
),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -50,6 +50,7 @@ import openai
|
|||
import pytest
|
||||
from e2e_config import REQUEST_TIMEOUT, unique_marker
|
||||
from e2e_http import unwrap
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from management.management_client import ManagementClient, build_client
|
||||
from models import KeyGenerateBody, KeyGenerateResponse, LiteLLMParamsBody, TeamNewBody, UserNewBody
|
||||
|
|
@ -172,6 +173,15 @@ def _assert_file_round_trip(client: OpenAI, native_id: str, marker: str) -> None
|
|||
|
||||
class TestAzureContainerFiles:
|
||||
@pytest.mark.covers("llm.responses.azure_openai.code_interpreter.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CONTAINERS,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_service_account_key_reads_container_file_by_native_id(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -187,6 +197,15 @@ class TestAzureContainerFiles:
|
|||
_assert_file_round_trip(client, native_id, marker)
|
||||
|
||||
@pytest.mark.covers("llm.responses.azure_openai.code_interpreter.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CONTAINERS,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_service_account_key_reads_container_file_created_by_a_streamed_response(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -202,6 +221,13 @@ class TestAzureContainerFiles:
|
|||
|
||||
|
||||
class TestOpenAIContainerFiles:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CONTAINERS,
|
||||
providers=(Provider.OPENAI,),
|
||||
)
|
||||
)
|
||||
def test_container_file_lifecycle_through_the_gateway(self, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
client: Final = sdk.openai(resources.key())
|
||||
marker: Final = unique_marker()
|
||||
|
|
|
|||
|
|
@ -3,10 +3,12 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import CredentialCreateBody, LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -14,9 +16,20 @@ from sdk_clients import NO_PROXY_CACHE, SdkClients
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
CLAUDE_BACKEND: Final = "anthropic/claude-haiku-4-5"
|
||||
|
||||
|
||||
class TestCredentialBackedMessages:
|
||||
@pytest.mark.covers("mgmt.credential.new.serves_request")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(CLAUDE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_credential_backed_messages(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
marker = unique_marker()
|
||||
credential_name = f"e2e-cred-{marker}"
|
||||
|
|
@ -35,7 +48,7 @@ class TestCredentialBackedMessages:
|
|||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="anthropic/claude-haiku-4-5",
|
||||
model=CLAUDE_BACKEND,
|
||||
litellm_credential_name=credential_name,
|
||||
),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ import pytest
|
|||
from pydantic import BaseModel, RootModel
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from proxy_client import ProxyClient
|
||||
from e2e_http import Success, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
|
|
@ -148,6 +149,15 @@ def _poll_breakdown_row(proxy: ProxyClient, key: str, response_id: str | None) -
|
|||
|
||||
|
||||
class TestCustomPricing:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.COST_MAP,
|
||||
route=Route.SPEND_REPORTING,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(BACKEND_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_custom_pricing_is_billed_at_configured_rate(
|
||||
self,
|
||||
proxy: ProxyClient,
|
||||
|
|
@ -193,6 +203,12 @@ class TestCustomPricing:
|
|||
f"= {completion * CUSTOM_OUTPUT_RATE}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.COST_MAP,
|
||||
route=Route.MODEL_MANAGEMENT,
|
||||
)
|
||||
)
|
||||
def test_model_info_reports_custom_pricing(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -208,6 +224,12 @@ class TestCustomPricing:
|
|||
f"{entry.litellm_params.output_cost_per_token} != configured {CUSTOM_OUTPUT_RATE}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.COST_MAP,
|
||||
route=Route.MODEL_MANAGEMENT,
|
||||
)
|
||||
)
|
||||
def test_custom_pricing_is_isolated_from_sibling_deployment(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ from __future__ import annotations
|
|||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from e2e_http import unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody, ThinkingParam
|
||||
|
|
@ -49,6 +50,16 @@ def _reasoning_content(response: ChatResponse) -> str | None:
|
|||
|
||||
|
||||
class TestDeepSeekReasoningDisable:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.DEEPSEEK,),
|
||||
models=(REASONER,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_reasoner_returns_reasoning_by_default(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -71,6 +82,16 @@ class TestDeepSeekReasoningDisable:
|
|||
f"disable param, so the disable assertions below can't be trusted: {response}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.DEEPSEEK,),
|
||||
models=(REASONER,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_reasoning_effort_none_disables_reasoning(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -93,6 +114,16 @@ class TestDeepSeekReasoningDisable:
|
|||
f"is still present: {response}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.DEEPSEEK,),
|
||||
models=(REASONER,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_thinking_disabled_disables_reasoning(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -16,6 +16,7 @@ from typing import Final
|
|||
import pytest
|
||||
from e2e_config import provider_edge_base, unique_marker
|
||||
from e2e_http import assert_client_error
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -24,6 +25,10 @@ from sdk_clients import NO_PROXY_CACHE, SdkClients, response_header
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
OPENAI_EMBEDDING: Final = "openai/text-embedding-3-small"
|
||||
BEDROCK_TITAN_EMBEDDING: Final = "bedrock/amazon.titan-embed-text-v2:0"
|
||||
COHERE_EMBEDDING: Final = "cohere/embed-v4.0"
|
||||
MISTRAL_EMBEDDING: Final = "mistral/mistral-embed"
|
||||
VERTEX_TEXT_EMBEDDING: Final = "vertex_ai/text-embedding-005"
|
||||
VERTEX_MULTIMODAL_EMBEDDING: Final = "vertex_ai/multimodalembedding@001"
|
||||
TOKENS_TEXT: Final = "The quick brown fox jumps over the lazy dog"
|
||||
|
|
@ -45,7 +50,7 @@ def _cosine(left: list[float], right: list[float]) -> float:
|
|||
|
||||
def _titan_params() -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(
|
||||
model="bedrock/amazon.titan-embed-text-v2:0",
|
||||
model=BEDROCK_TITAN_EMBEDDING,
|
||||
aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID",
|
||||
aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY",
|
||||
aws_region_name="os.environ/AWS_REGION",
|
||||
|
|
@ -62,7 +67,7 @@ def _openai_embeddings_params() -> LiteLLMParamsBody:
|
|||
Vertex stay live: SigV4 signs the Host header, and neither has an edge mount."""
|
||||
base = provider_edge_base("openai")
|
||||
return LiteLLMParamsBody(
|
||||
model="openai/text-embedding-3-small",
|
||||
model=OPENAI_EMBEDDING,
|
||||
api_key="os.environ/OPENAI_API_KEY",
|
||||
api_base=None if base is None else f"{base}/v1",
|
||||
)
|
||||
|
|
@ -97,10 +102,28 @@ def _assert_embedding_vector(
|
|||
class TestEmbeddingsEndpoint:
|
||||
@pytest.mark.replayable
|
||||
@pytest.mark.covers("llm.embeddings.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_embeddings_returns_vector(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
_assert_embedding_vector(proxy, resources, sdk, "e2e-embeddings", _openai_embeddings_params())
|
||||
|
||||
@pytest.mark.covers("llm.embeddings.bedrock.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_TITAN_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_embeddings_returns_vector(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -113,6 +136,15 @@ class TestEmbeddingsEndpoint:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.embeddings.cohere.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.COHERE,),
|
||||
models=(COHERE_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_cohere_embeddings_returns_vector(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -121,10 +153,19 @@ class TestEmbeddingsEndpoint:
|
|||
resources,
|
||||
sdk,
|
||||
"e2e-embeddings-cohere",
|
||||
LiteLLMParamsBody(model="cohere/embed-v4.0", api_key="os.environ/COHERE_API_KEY"),
|
||||
LiteLLMParamsBody(model=COHERE_EMBEDDING, api_key="os.environ/COHERE_API_KEY"),
|
||||
)
|
||||
|
||||
@pytest.mark.covers("llm.embeddings.vertex.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_TEXT_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_embeddings_returns_vector(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -134,12 +175,21 @@ class TestEmbeddingsEndpoint:
|
|||
sdk,
|
||||
"e2e-embeddings-vertex",
|
||||
LiteLLMParamsBody(
|
||||
model="vertex_ai/text-embedding-005",
|
||||
model=VERTEX_TEXT_EMBEDDING,
|
||||
vertex_project="os.environ/VERTEXAI_PROJECT",
|
||||
vertex_location="us-central1",
|
||||
),
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.MISTRAL,),
|
||||
models=(MISTRAL_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_mistral_embeddings_returns_vector(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -148,10 +198,19 @@ class TestEmbeddingsEndpoint:
|
|||
resources,
|
||||
sdk,
|
||||
"e2e-embeddings-mistral",
|
||||
LiteLLMParamsBody(model="mistral/mistral-embed", api_key="os.environ/MISTRAL_API_KEY"),
|
||||
LiteLLMParamsBody(model=MISTRAL_EMBEDDING, api_key="os.environ/MISTRAL_API_KEY"),
|
||||
)
|
||||
|
||||
@pytest.mark.covers("llm.embeddings.vertex.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_TEXT_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_embeddings_honor_requested_dimensions(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -166,6 +225,15 @@ class TestEmbeddingsEndpoint:
|
|||
assert len(embeddings.data[0].embedding) == 8, f"dimensions=8 was not honored: {embeddings!r}"
|
||||
assert embeddings.usage.prompt_tokens > 0, f"vertex embeddings reported no prompt usage: {embeddings.usage!r}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_MULTIMODAL_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_multimodal_embeddings_honor_dimensions_and_are_costed(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -181,6 +249,15 @@ class TestEmbeddingsEndpoint:
|
|||
cost = response_header(raw.headers, "x-litellm-response-cost")
|
||||
assert cost is not None and float(cost) > 0, f"multimodal embedding was not costed: {cost!r}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_TITAN_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_titan_embeds_token_array_input_as_its_decoded_text(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -199,6 +276,15 @@ class TestEmbeddingsEndpoint:
|
|||
|
||||
@pytest.mark.replayable
|
||||
@pytest.mark.covers("llm.embeddings.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_EMBEDDING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_array_input_returns_vectors(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model, key = _register(proxy, resources, "e2e-embeddings-array", _openai_embeddings_params())
|
||||
embeddings = sdk.openai(key).embeddings.create(
|
||||
|
|
@ -208,6 +294,12 @@ class TestEmbeddingsEndpoint:
|
|||
|
||||
@pytest.mark.replayable
|
||||
@pytest.mark.covers("llm.embeddings.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
)
|
||||
)
|
||||
def test_missing_model_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -219,6 +311,12 @@ class TestEmbeddingsEndpoint:
|
|||
|
||||
@pytest.mark.replayable
|
||||
@pytest.mark.covers("llm.embeddings.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.EMBEDDINGS,
|
||||
)
|
||||
)
|
||||
def test_missing_input_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources, "e2e-embeddings-missin", _openai_embeddings_params())
|
||||
result = proxy.transport.send(
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ from __future__ import annotations
|
|||
|
||||
import pytest
|
||||
from e2e_http import NoBody, Success, UnknownApiError, assert_client_error
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from proxy_client import ProxyClient
|
||||
from pydantic import BaseModel
|
||||
|
|
@ -28,6 +29,13 @@ class BatchObject(BaseModel):
|
|||
|
||||
class TestFilesBatchesContract:
|
||||
@pytest.mark.covers("llm.files.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.FILES,
|
||||
mode=Mode.BATCH,
|
||||
)
|
||||
)
|
||||
def test_upload_without_purpose_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
result = proxy.transport.upload(
|
||||
|
|
@ -47,6 +55,13 @@ class TestFilesBatchesContract:
|
|||
pytest.fail(f"upload without purpose expected 4xx, got {other!r}")
|
||||
|
||||
@pytest.mark.covers("llm.batches.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.BATCHES,
|
||||
mode=Mode.BATCH,
|
||||
)
|
||||
)
|
||||
def test_create_batch_missing_input_file_id_returns_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -59,6 +74,14 @@ class TestFilesBatchesContract:
|
|||
assert_client_error(result, "batch missing input_file_id")
|
||||
|
||||
@pytest.mark.covers("llm.batches.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.BATCHES,
|
||||
providers=(Provider.OPENAI,),
|
||||
mode=Mode.BATCH,
|
||||
)
|
||||
)
|
||||
def test_retrieve_invalid_batch_id_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
result = proxy.transport.get(
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ from typing import Literal
|
|||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import StreamingResponse, require_successful_call
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -85,6 +86,15 @@ def _streamed_text(result: StreamingResponse) -> str:
|
|||
|
||||
class TestGoogleNativeGenerateContent:
|
||||
@pytest.mark.covers("llm.google_native.gemini.basic.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.GOOGLE_GENAI,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(UPSTREAM_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_generate_content_returns_response_cost_header(
|
||||
self,
|
||||
proxy: ProxyClient,
|
||||
|
|
@ -104,6 +114,15 @@ class TestGoogleNativeGenerateContent:
|
|||
assert result.response_cost > 0, f"x-litellm-response-cost must be a real cost, got {result.response_cost}"
|
||||
|
||||
@pytest.mark.covers("llm.google_native.gemini.basic.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.GOOGLE_GENAI,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(UPSTREAM_MODEL,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_stream_generate_content_frames_sse_the_way_google_sdks_expect(
|
||||
self,
|
||||
proxy: ProxyClient,
|
||||
|
|
|
|||
|
|
@ -11,10 +11,12 @@ as the `image` part, not a JSON body. The fixture image is a small generated
|
|||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
from typing import Final
|
||||
|
||||
import openai
|
||||
import pytest
|
||||
from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS, unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -22,6 +24,8 @@ from sdk_clients import SdkClients
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
IMAGE_EDIT_BACKEND: Final = "openai/gpt-image-1"
|
||||
|
||||
_TEST_PNG = base64.b64decode(
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAEAAAABACAIAAAAlC+aJAAAAS0lEQVR42u3PMQ0AAAwDoPo3"
|
||||
"3UrYvQQckD4XAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEB"
|
||||
|
|
@ -33,7 +37,7 @@ def _register_image_model(proxy: ProxyClient, resources: ResourceManager) -> tup
|
|||
model = f"e2e-image-edit-{unique_marker()}"
|
||||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(model="openai/gpt-image-1", api_key="os.environ/OPENAI_API_KEY"),
|
||||
LiteLLMParamsBody(model=IMAGE_EDIT_BACKEND, api_key="os.environ/OPENAI_API_KEY"),
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
return model, resources.key()
|
||||
|
|
@ -49,6 +53,15 @@ def _assert_client_error(error: openai.APIStatusError, context: str) -> None:
|
|||
|
||||
class TestImageEdit:
|
||||
@pytest.mark.covers("llm.images_edits.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(IMAGE_EDIT_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_image_edit_returns_image(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model, key = _register_image_model(proxy, resources)
|
||||
client = sdk.openai(key)
|
||||
|
|
@ -64,6 +77,15 @@ class TestImageEdit:
|
|||
assert first.b64_json or first.url, f"edited image has neither b64_json nor url: {first!r}"
|
||||
|
||||
@pytest.mark.covers("llm.images_edits.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(IMAGE_EDIT_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_empty_prompt_returns_error(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model, key = _register_image_model(proxy, resources)
|
||||
client = sdk.openai(key)
|
||||
|
|
@ -73,6 +95,15 @@ class TestImageEdit:
|
|||
_assert_client_error(raised.value, "empty image-edit prompt")
|
||||
|
||||
@pytest.mark.covers("llm.images_edits.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(IMAGE_EDIT_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_empty_image_returns_error(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model, key = _register_image_model(proxy, resources)
|
||||
client = sdk.openai(key)
|
||||
|
|
|
|||
|
|
@ -8,9 +8,12 @@ from litellm-regression-tests/tests/test_inference_endpoints.py.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import assert_client_error
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from openai.types import ImagesResponse
|
||||
|
|
@ -20,6 +23,9 @@ from sdk_clients import SdkClients
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
OPENAI_IMAGE_BACKEND: Final = "openai/gpt-image-1-mini"
|
||||
BEDROCK_IMAGE_BACKEND: Final = "bedrock/amazon.nova-canvas-v1:0"
|
||||
|
||||
|
||||
class _OptionalImageBody(BaseModel):
|
||||
model: str | None = None
|
||||
|
|
@ -47,12 +53,21 @@ def _register_openai_image(proxy: ProxyClient, resources: ResourceManager) -> tu
|
|||
proxy,
|
||||
resources,
|
||||
"e2e-image",
|
||||
LiteLLMParamsBody(model="openai/gpt-image-1-mini", api_key="os.environ/OPENAI_API_KEY"),
|
||||
LiteLLMParamsBody(model=OPENAI_IMAGE_BACKEND, api_key="os.environ/OPENAI_API_KEY"),
|
||||
)
|
||||
|
||||
|
||||
class TestImageGeneration:
|
||||
@pytest.mark.covers("llm.images_generations.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_IMAGE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_image_generation_returns_image(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -61,6 +76,15 @@ class TestImageGeneration:
|
|||
_assert_image_returned(images)
|
||||
|
||||
@pytest.mark.covers("llm.images_generations.bedrock.basic.nonstream.works", exercised_on=["images_generations"])
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_IMAGE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_image_generation_returns_image(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -69,7 +93,7 @@ class TestImageGeneration:
|
|||
resources,
|
||||
"e2e-bedrock-image",
|
||||
LiteLLMParamsBody(
|
||||
model="bedrock/amazon.nova-canvas-v1:0",
|
||||
model=BEDROCK_IMAGE_BACKEND,
|
||||
aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID",
|
||||
aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY",
|
||||
aws_region_name="os.environ/AWS_REGION",
|
||||
|
|
@ -80,6 +104,12 @@ class TestImageGeneration:
|
|||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /v1/images/generations 500s (aimage_generation TypeError) on missing prompt instead of 400")
|
||||
@pytest.mark.covers("llm.images_generations.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
)
|
||||
)
|
||||
def test_missing_prompt_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_openai_image(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -90,6 +120,15 @@ class TestImageGeneration:
|
|||
assert_client_error(result, "images missing prompt")
|
||||
|
||||
@pytest.mark.covers("llm.images_generations.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_IMAGE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_empty_prompt_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_openai_image(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -100,6 +139,15 @@ class TestImageGeneration:
|
|||
assert_client_error(result, "images empty prompt")
|
||||
|
||||
@pytest.mark.covers("llm.images_generations.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_IMAGE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invalid_size_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_openai_image(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -110,6 +158,15 @@ class TestImageGeneration:
|
|||
assert_client_error(result, "images invalid size")
|
||||
|
||||
@pytest.mark.covers("llm.images_generations.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.IMAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_IMAGE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_invalid_n_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register_openai_image(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ import pytest
|
|||
from anthropic.types import RawMessageStreamEvent, ToolParam
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -56,6 +57,15 @@ class TestAzureFoundryMessages:
|
|||
return model
|
||||
|
||||
@pytest.mark.covers("llm.messages.azure_foundry.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(AZURE_FOUNDRY_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_basic_nonstream(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model = self._register(proxy, resources)
|
||||
client = sdk.anthropic(resources.key(models=[model]))
|
||||
|
|
@ -71,6 +81,15 @@ class TestAzureFoundryMessages:
|
|||
assert text.strip(), f"/v1/messages returned no text: {message.content!r}"
|
||||
|
||||
@pytest.mark.covers("llm.messages.azure_foundry.basic.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(AZURE_FOUNDRY_MODEL,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_basic_stream(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model = self._register(proxy, resources)
|
||||
client = sdk.anthropic(resources.key(models=[model]))
|
||||
|
|
@ -85,6 +104,16 @@ class TestAzureFoundryMessages:
|
|||
_assert_streamed_ok([event.type for event in stream])
|
||||
|
||||
@pytest.mark.covers("llm.messages.azure_foundry.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(AZURE_FOUNDRY_MODEL,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_tool_use_nonstream(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model = self._register(proxy, resources)
|
||||
client = sdk.anthropic(resources.key(models=[model]))
|
||||
|
|
@ -102,6 +131,16 @@ class TestAzureFoundryMessages:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.messages.azure_foundry.tool_use.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(AZURE_FOUNDRY_MODEL,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_tool_use_stream(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model = self._register(proxy, resources)
|
||||
client = sdk.anthropic(resources.key(models=[model]))
|
||||
|
|
@ -122,6 +161,16 @@ class TestAzureFoundryMessages:
|
|||
), "stream carried no tool_use block"
|
||||
assert "message_stop" in event_types, "stream never reached message_stop"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(AZURE_FOUNDRY_MODEL,),
|
||||
capabilities=(Capability.RESPONSE_SCHEMA,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_output_format_returns_schema_json(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ from typing import Final
|
|||
import pytest
|
||||
from anthropic.types import RawContentBlockDeltaEvent, RawMessageDeltaEvent, TextBlock, TextDelta
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -33,6 +34,16 @@ def _register(proxy: ProxyClient, resources: ResourceManager, backend: str) -> s
|
|||
|
||||
|
||||
class TestBedrockMessages:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(CONVERSE_CLAUDE_BACKEND,),
|
||||
capabilities=(Capability.RESPONSE_SCHEMA,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_converse_output_format_returns_schema_json_text(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -50,6 +61,15 @@ class TestBedrockMessages:
|
|||
assert_sentiment_json("".join(texts))
|
||||
|
||||
@pytest.mark.covers("llm.messages.bedrock_converse.basic.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(NOVA_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_nova_stream_relays_text_usage_and_stop(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from typing import Final
|
|||
|
||||
import anthropic
|
||||
import pytest
|
||||
from _pytest.mark.structures import ParameterSet
|
||||
from anthropic import Anthropic
|
||||
from anthropic.types import (
|
||||
InputJSONDelta,
|
||||
|
|
@ -42,6 +43,7 @@ from e2e_config import (
|
|||
unique_marker,
|
||||
)
|
||||
from e2e_http import assert_client_error
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import AnthropicErrorEvent, AnthropicMessagesBody, ChatMessage, LiteLLMParamsBody, SpendLogRow
|
||||
from provider_edge import EDGE_MOUNTS, LiveEdge, RunningEdge, StreamCut, start_provider_edge
|
||||
|
|
@ -61,6 +63,7 @@ class _OptionalMessagesBody(BaseModel):
|
|||
|
||||
|
||||
ANTHROPIC_BACKEND = "anthropic/claude-haiku-4-5"
|
||||
OPENAI_BRIDGE_BACKEND: Final = "openai/gpt-5.6"
|
||||
|
||||
WEATHER_TOOL: ToolParam = {
|
||||
"name": "get_weather",
|
||||
|
|
@ -110,6 +113,15 @@ def _user_turn(text: str) -> MessageParam:
|
|||
|
||||
class TestAnthropicMessages:
|
||||
@pytest.mark.covers("llm.messages.anthropic.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_messages_returns_completion(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
client = sdk.anthropic(key)
|
||||
|
|
@ -121,6 +133,15 @@ class TestAnthropicMessages:
|
|||
assert _text(message).strip(), f"/v1/messages returned no text: {message.content!r}"
|
||||
|
||||
@pytest.mark.covers("llm.messages.anthropic.basic.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_messages_logs_cost_matching_the_response_header(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -169,6 +190,15 @@ class TestAnthropicMessages:
|
|||
|
||||
@pytest.mark.covers("llm.messages.anthropic.basic.stream.works")
|
||||
@pytest.mark.provider_live
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_messages_streams_completion(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
"""Edge-wired like its non-streaming siblings, so record and replay both
|
||||
carry the streamed response.
|
||||
|
|
@ -226,6 +256,16 @@ class TestAnthropicMessages:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.messages.anthropic.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_messages_tool_use(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
client = sdk.anthropic(key)
|
||||
|
|
@ -243,6 +283,16 @@ class TestAnthropicMessages:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.messages.anthropic.structured_output.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.RESPONSE_SCHEMA,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_messages_output_format_returns_schema_json(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -261,6 +311,14 @@ class TestAnthropicMessages:
|
|||
reason="stage red: product gap, /v1/messages 500s (anthropic_messages TypeError) on missing messages instead of 400"
|
||||
)
|
||||
@pytest.mark.covers("llm.messages.anthropic.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(),
|
||||
models=(),
|
||||
)
|
||||
)
|
||||
def test_missing_messages_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -274,6 +332,14 @@ class TestAnthropicMessages:
|
|||
reason="stage red: product gap, /v1/messages 500s (anthropic_messages TypeError) on missing max_tokens instead of 400"
|
||||
)
|
||||
@pytest.mark.covers("llm.messages.anthropic.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(),
|
||||
models=(),
|
||||
)
|
||||
)
|
||||
def test_missing_max_tokens_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -284,6 +350,14 @@ class TestAnthropicMessages:
|
|||
assert_client_error(result, "messages missing max_tokens")
|
||||
|
||||
@pytest.mark.covers("llm.messages.anthropic.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(),
|
||||
models=(),
|
||||
)
|
||||
)
|
||||
def test_missing_model_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
_, key = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -364,9 +438,26 @@ def _request_tool(client: Anthropic, model: str, question: MessageParam, tool: T
|
|||
return blocks[0]
|
||||
|
||||
|
||||
def _openai_bridge_subject(mode: Mode) -> Subject:
|
||||
return Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BRIDGE_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=mode,
|
||||
)
|
||||
|
||||
|
||||
class TestOpenAIMessagesToolContinuation:
|
||||
@pytest.mark.provider_live
|
||||
@pytest.mark.parametrize("stream", [True, False], ids=["stream", "nonstream"])
|
||||
@pytest.mark.parametrize(
|
||||
"stream",
|
||||
[
|
||||
pytest.param(stream, marks=meta(_openai_bridge_subject(mode)), id=name)
|
||||
for stream, name, mode in ((True, "stream", Mode.STREAM), (False, "nonstream", Mode.NONSTREAM))
|
||||
],
|
||||
)
|
||||
def test_required_tool_arguments_and_correlated_result(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, stream: bool
|
||||
) -> None:
|
||||
|
|
@ -375,7 +466,7 @@ class TestOpenAIMessagesToolContinuation:
|
|||
model_id: Final = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/gpt-5.6", api_key="os.environ/OPENAI_API_KEY", api_base=f"{base}/v1" if base else None
|
||||
model=OPENAI_BRIDGE_BACKEND, api_key="os.environ/OPENAI_API_KEY", api_base=f"{base}/v1" if base else None
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
|
|
@ -478,6 +569,30 @@ _DROPPED_BEFORE_FIRST_BYTE: Final[tuple[tuple[str, _CutRegistration, StreamCut],
|
|||
)
|
||||
|
||||
|
||||
_CUT_SUBJECTS: Final[MappingProxyType[_CutRegistration, Subject]] = MappingProxyType(
|
||||
{
|
||||
_register_cut_bedrock: Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
),
|
||||
_register_cut_anthropic: Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _cut_params(cases: tuple[tuple[str, _CutRegistration, StreamCut], ...]) -> list[ParameterSet]:
|
||||
return [pytest.param(register, cut, id=name, marks=meta(_CUT_SUBJECTS[register])) for name, register, cut in cases]
|
||||
|
||||
|
||||
def _payload(frame: str) -> JsonValue | None:
|
||||
try:
|
||||
return _FRAME_PAYLOAD.validate_json(frame)
|
||||
|
|
@ -495,7 +610,7 @@ def _bare_error_frame(frame: str) -> bool:
|
|||
class TestMessagesUpstreamStreamFailure:
|
||||
@pytest.mark.covers("llm.messages.anthropic.upstream_stream_failure.stream.error_event")
|
||||
@pytest.mark.parametrize(
|
||||
("register", "cut"), [case[1:] for case in _DROPPED_UPSTREAMS], ids=[case[0] for case in _DROPPED_UPSTREAMS]
|
||||
("register", "cut"), _cut_params(_DROPPED_UPSTREAMS)
|
||||
)
|
||||
def test_interrupted_upstream_stream_raises_in_the_anthropic_sdk(
|
||||
self,
|
||||
|
|
@ -533,7 +648,7 @@ class TestMessagesUpstreamStreamFailure:
|
|||
|
||||
@pytest.mark.covers("llm.messages.anthropic.upstream_stream_failure.stream.error_event")
|
||||
@pytest.mark.parametrize(
|
||||
("register", "cut"), [case[1:] for case in _DROPPED_UPSTREAMS], ids=[case[0] for case in _DROPPED_UPSTREAMS]
|
||||
("register", "cut"), _cut_params(_DROPPED_UPSTREAMS)
|
||||
)
|
||||
def test_interrupted_upstream_stream_is_an_anthropic_error_event(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, register: _CutRegistration, cut: StreamCut
|
||||
|
|
@ -585,11 +700,7 @@ class TestMessagesUpstreamStreamFailure:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.messages.anthropic.upstream_stream_failure.stream.error_status")
|
||||
@pytest.mark.parametrize(
|
||||
("register", "cut"),
|
||||
[case[1:] for case in _DROPPED_BEFORE_FIRST_BYTE],
|
||||
ids=[case[0] for case in _DROPPED_BEFORE_FIRST_BYTE],
|
||||
)
|
||||
@pytest.mark.parametrize(("register", "cut"), _cut_params(_DROPPED_BEFORE_FIRST_BYTE))
|
||||
def test_upstream_that_hangs_up_before_the_first_byte_raises_with_its_status_in_the_anthropic_sdk(
|
||||
self,
|
||||
proxy: ProxyClient,
|
||||
|
|
@ -622,11 +733,7 @@ class TestMessagesUpstreamStreamFailure:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.messages.anthropic.upstream_stream_failure.stream.error_status")
|
||||
@pytest.mark.parametrize(
|
||||
("register", "cut"),
|
||||
[case[1:] for case in _DROPPED_BEFORE_FIRST_BYTE],
|
||||
ids=[case[0] for case in _DROPPED_BEFORE_FIRST_BYTE],
|
||||
)
|
||||
@pytest.mark.parametrize(("register", "cut"), _cut_params(_DROPPED_BEFORE_FIRST_BYTE))
|
||||
def test_upstream_that_hangs_up_before_the_first_byte_is_a_json_error_with_its_status(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, register: _CutRegistration, cut: StreamCut
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -34,6 +34,7 @@ import pytest
|
|||
from anthropic import Anthropic
|
||||
from anthropic.types import Message, MessageParam, TextBlockParam
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -208,6 +209,16 @@ class TestBedrockInvokeMidConversationSystem:
|
|||
"llm.messages.bedrock_invoke.mid_conversation_system.nonstream.cache_hit",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(FLAGGED_INVOKE_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -234,6 +245,16 @@ class TestBedrockInvokeMidConversationSystem:
|
|||
"llm.messages.bedrock_invoke.mid_conversation_system.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(UNFLAGGED_INVOKE_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_unflagged_model_converts_system_reminder_and_succeeds(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -41,6 +41,7 @@ import pytest
|
|||
from anthropic import Anthropic
|
||||
from anthropic.types import Message, MessageParam, TextBlockParam
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -290,6 +291,16 @@ class TestAzureFoundryMidConversationSystem:
|
|||
"llm.messages.azure_foundry.mid_conversation_system.nonstream.cache_hit",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(FLAGGED_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -299,6 +310,16 @@ class TestAzureFoundryMidConversationSystem:
|
|||
"llm.messages.azure_foundry.mid_conversation_system.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.AZURE_AI,),
|
||||
models=(UNFLAGGED_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_unflagged_model_converts_system_reminder_and_succeeds(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -322,6 +343,16 @@ class TestVertexMidConversationSystem:
|
|||
"llm.messages.vertex.mid_conversation_system.nonstream.cache_hit",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(FLAGGED_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -333,6 +364,16 @@ class TestVertexMidConversationSystem:
|
|||
"llm.messages.vertex.mid_conversation_system.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(UNFLAGGED_MODEL,),
|
||||
capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_unflagged_model_converts_system_reminder_and_succeeds(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -9,9 +9,12 @@ negative stays on the shared transport, since the SDK refuses to send it.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import assert_client_error
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from openai.types import Moderation
|
||||
|
|
@ -21,6 +24,7 @@ from sdk_clients import SdkClients
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
OPENAI_MODERATION_BACKEND: Final = "openai/omni-moderation-latest"
|
||||
VIOLENT_TEXT = "I am going to find you and kill you, and I will hurt everyone you love."
|
||||
BENIGN_TEXT = "I enjoyed the sunny afternoon and a relaxing walk in the park today."
|
||||
|
||||
|
|
@ -35,7 +39,7 @@ def _register_moderation_model(proxy: ProxyClient, resources: ResourceManager) -
|
|||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/omni-moderation-latest", api_key="os.environ/OPENAI_API_KEY"
|
||||
model=OPENAI_MODERATION_BACKEND, api_key="os.environ/OPENAI_API_KEY"
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
|
|
@ -52,6 +56,15 @@ def _flagged_categories(item: Moderation) -> tuple[str, ...]:
|
|||
|
||||
class TestModerations:
|
||||
@pytest.mark.covers("llm.moderations.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MODERATIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MODERATION_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_moderations_flags_violent_content(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -64,6 +77,15 @@ class TestModerations:
|
|||
assert item.flagged, f"violent text was not flagged: {item!r}"
|
||||
assert _flagged_categories(item), f"flagged result reported no true category: {item!r}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MODERATIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MODERATION_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_moderations_passes_benign_content(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -79,6 +101,14 @@ class TestModerations:
|
|||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /v1/moderations 500s (KeyError 'input') on missing input instead of 400")
|
||||
@pytest.mark.covers("llm.moderations.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MODERATIONS,
|
||||
providers=(),
|
||||
models=(),
|
||||
)
|
||||
)
|
||||
def test_missing_input_returns_error(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -25,6 +25,7 @@ from typing import Final, Protocol
|
|||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from e2e_http import (
|
||||
PROVIDER_RATE_LIMIT_ATTEMPTS,
|
||||
RateLimitedError,
|
||||
|
|
@ -60,6 +61,13 @@ TEST_IMAGE_URL = (
|
|||
)
|
||||
|
||||
|
||||
MISTRAL_OCR_MODEL: Final = "mistral/mistral-ocr-latest"
|
||||
AZURE_AI_OCR_MODEL: Final = "azure_ai/mistral-document-ai-2512"
|
||||
AZURE_DOC_INTELLIGENCE_MODEL: Final = "azure_ai/doc-intelligence/prebuilt-layout"
|
||||
VERTEX_OCR_MODEL: Final = "vertex_ai/mistral-ocr-2505"
|
||||
COHERE_OCR_MODEL: Final = "cohere/parse-v5.0"
|
||||
|
||||
|
||||
class OcrProvider(Protocol):
|
||||
"""One OCR provider's deployment config: its model id plus the os.environ/*
|
||||
credential references the proxy resolves at call time. Each provider owns which
|
||||
|
|
@ -70,7 +78,7 @@ class OcrProvider(Protocol):
|
|||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MistralOcr:
|
||||
model: str = "mistral/mistral-ocr-latest"
|
||||
model: str = MISTRAL_OCR_MODEL
|
||||
|
||||
def litellm_params(self) -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=self.model, api_key="os.environ/MISTRAL_API_KEY")
|
||||
|
|
@ -96,7 +104,7 @@ class AzureDocIntelligenceOcr:
|
|||
AZURE_DOCUMENT_INTELLIGENCE_API_KEY, which the OCR config resolves from the
|
||||
doc-intelligence model name when api_base/api_key are left unset."""
|
||||
|
||||
model: str = "azure_ai/doc-intelligence/prebuilt-layout"
|
||||
model: str = AZURE_DOC_INTELLIGENCE_MODEL
|
||||
|
||||
def litellm_params(self) -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=self.model)
|
||||
|
|
@ -120,7 +128,7 @@ class VertexOcr:
|
|||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class CohereOcr:
|
||||
model: str = "cohere/parse-v5.0"
|
||||
model: str = COHERE_OCR_MODEL
|
||||
|
||||
def litellm_params(self) -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=self.model, api_key="os.environ/COHERE_API_KEY")
|
||||
|
|
@ -141,7 +149,7 @@ RUST_OCR_CASES: tuple[_OcrCase, ...] = (
|
|||
),
|
||||
_OcrCase(
|
||||
"azure-ai",
|
||||
AzureAiOcr("azure_ai/mistral-document-ai-2512"),
|
||||
AzureAiOcr(AZURE_AI_OCR_MODEL),
|
||||
OcrDocument(type="document_url", document_url=TEST_PDF_URL),
|
||||
),
|
||||
_OcrCase(
|
||||
|
|
@ -151,13 +159,11 @@ RUST_OCR_CASES: tuple[_OcrCase, ...] = (
|
|||
),
|
||||
_OcrCase(
|
||||
"vertex-mistral",
|
||||
VertexOcr("vertex_ai/mistral-ocr-2505", "us-central1"),
|
||||
VertexOcr(VERTEX_OCR_MODEL, "us-central1"),
|
||||
OcrDocument(type="document_url", document_url=TEST_PDF_URL),
|
||||
),
|
||||
)
|
||||
|
||||
_CASE_IDS = tuple(case.suffix for case in RUST_OCR_CASES)
|
||||
|
||||
PDF_TEXT: Final = "test pdf file"
|
||||
IMAGE_TEXT: Final = "litellm"
|
||||
PDF_DOCUMENT: Final = OcrDocument(type="document_url", document_url=TEST_PDF_URL)
|
||||
|
|
@ -175,14 +181,37 @@ class _OcrContentCase:
|
|||
OCR_CONTENT_CASES: Final = (
|
||||
_OcrContentCase("mistral-pdf", MistralOcr(), PDF_DOCUMENT, PDF_TEXT),
|
||||
_OcrContentCase("mistral-image", MistralOcr(), IMAGE_DOCUMENT, IMAGE_TEXT),
|
||||
_OcrContentCase("azure-ai-image", AzureAiOcr("azure_ai/mistral-document-ai-2512"), IMAGE_DOCUMENT, IMAGE_TEXT),
|
||||
_OcrContentCase("azure-ai-image", AzureAiOcr(AZURE_AI_OCR_MODEL), IMAGE_DOCUMENT, IMAGE_TEXT),
|
||||
_OcrContentCase(
|
||||
"vertex-mistral-image", VertexOcr("vertex_ai/mistral-ocr-2505", "us-central1"), IMAGE_DOCUMENT, IMAGE_TEXT
|
||||
"vertex-mistral-image", VertexOcr(VERTEX_OCR_MODEL, "us-central1"), IMAGE_DOCUMENT, IMAGE_TEXT
|
||||
),
|
||||
_OcrContentCase("cohere-image", CohereOcr(), IMAGE_DOCUMENT, IMAGE_TEXT),
|
||||
)
|
||||
|
||||
|
||||
def _ocr_subject(provider: OcrProvider) -> Subject:
|
||||
match provider:
|
||||
case MistralOcr():
|
||||
vendor, model = Provider.MISTRAL, MISTRAL_OCR_MODEL
|
||||
case AzureAiOcr():
|
||||
vendor, model = Provider.AZURE_AI, AZURE_AI_OCR_MODEL
|
||||
case AzureDocIntelligenceOcr():
|
||||
vendor, model = Provider.AZURE_AI, AZURE_DOC_INTELLIGENCE_MODEL
|
||||
case VertexOcr():
|
||||
vendor, model = Provider.VERTEX_AI, VERTEX_OCR_MODEL
|
||||
case CohereOcr():
|
||||
vendor, model = Provider.COHERE, COHERE_OCR_MODEL
|
||||
case _:
|
||||
raise TypeError(f"no OCR subject for {provider!r}")
|
||||
return Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.OCR,
|
||||
providers=(vendor,),
|
||||
models=(model,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
|
||||
|
||||
def _assert_ocr_document(response: OcrResponse) -> None:
|
||||
assert response.object == "ocr", f"expected object='ocr', got {response.object!r}"
|
||||
assert response.model, "response missing the resolved model name"
|
||||
|
|
@ -202,7 +231,10 @@ def _assert_provider_rate_limit_relayed(model: str, outcome: RateLimitedError) -
|
|||
|
||||
|
||||
class TestRustOcrGateway:
|
||||
@pytest.mark.parametrize("case", RUST_OCR_CASES, ids=_CASE_IDS)
|
||||
@pytest.mark.parametrize(
|
||||
"case",
|
||||
[pytest.param(case, marks=meta(_ocr_subject(case.provider)), id=case.suffix) for case in RUST_OCR_CASES],
|
||||
)
|
||||
def test_rust_ocr_response(self, proxy: ProxyClient, resources: ResourceManager, case: _OcrCase) -> None:
|
||||
model = f"rust-ocr-{case.suffix}-{unique_marker()}"
|
||||
model_id = proxy.create_model(model, case.provider.litellm_params())
|
||||
|
|
@ -219,6 +251,14 @@ class TestRustOcrGateway:
|
|||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /v1/ocr 500s (aocr TypeError) on missing document instead of 400")
|
||||
@pytest.mark.covers("llm.ocr.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.OCR,
|
||||
providers=(),
|
||||
models=(),
|
||||
)
|
||||
)
|
||||
def test_missing_document_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model = f"rust-ocr-val-{unique_marker()}"
|
||||
model_id = proxy.create_model(model, MistralOcr().litellm_params())
|
||||
|
|
@ -233,7 +273,13 @@ class TestRustOcrGateway:
|
|||
|
||||
|
||||
class TestOcrDocumentContent:
|
||||
@pytest.mark.parametrize("case", OCR_CONTENT_CASES, ids=tuple(case.suffix for case in OCR_CONTENT_CASES))
|
||||
@pytest.mark.parametrize(
|
||||
"case",
|
||||
[
|
||||
pytest.param(case, marks=meta(_ocr_subject(case.provider)), id=case.suffix)
|
||||
for case in OCR_CONTENT_CASES
|
||||
],
|
||||
)
|
||||
def test_ocr_reads_the_document_and_bills_its_pages(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, case: _OcrContentCase
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -12,10 +12,13 @@ A passthrough call returning non-2xx fails hard (never a skip); once it returns
|
|||
2xx, a missing or zero-cost SpendLogs row fails too.
|
||||
"""
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import CHEAP_OPENAI_MODEL, unique_marker
|
||||
from e2e_http import require_successful_call, unwrap
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatResponse, KeyGenerateBody, SpendLogRow
|
||||
from passthrough_client import (
|
||||
|
|
@ -29,6 +32,8 @@ from passthrough_client import (
|
|||
completed_responses_object,
|
||||
)
|
||||
|
||||
GEMINI_MODEL: Final = "gemini-2.5-flash"
|
||||
ANTHROPIC_PASSTHROUGH_MODEL: Final = "claude-haiku-4-5"
|
||||
EMBEDDING_MODEL = "text-embedding-3-small"
|
||||
REALTIME_MODEL = "gpt-realtime-2"
|
||||
|
||||
|
|
@ -57,12 +62,21 @@ def _fetch_cost_breakdown(client: PassthroughClient, request_id: str | None) ->
|
|||
# ---- Gemini passthrough ------------------------------------------------
|
||||
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_gemini_passthrough_nonstreaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
tag = f"e2e-passthrough-{unique_marker()}"
|
||||
result = client.gemini_generate(
|
||||
scoped_key, "gemini-2.5-flash", "Say hello in one word", tags=[tag, "gemini"]
|
||||
scoped_key, GEMINI_MODEL, "Say hello in one word", tags=[tag, "gemini"]
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
|
|
@ -73,6 +87,15 @@ def test_gemini_passthrough_nonstreaming_logs_cost(
|
|||
|
||||
|
||||
@pytest.mark.skip(reason="stage red: product gap, native passthrough returns no x-litellm-response-cost or x-ratelimit-* headers")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -82,7 +105,7 @@ def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_rout
|
|||
today, which makes native traffic invisible to the same tooling.
|
||||
"""
|
||||
result = client.gemini_generate(
|
||||
scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}"
|
||||
scoped_key, GEMINI_MODEL, f"Say hello in one word. {unique_marker()}"
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
|
|
@ -102,10 +125,19 @@ def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_rout
|
|||
)
|
||||
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_MODEL,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_gemini_passthrough_streaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.gemini_stream(scoped_key, "gemini-2.5-flash", "Count to five")
|
||||
result = client.gemini_stream(scoped_key, GEMINI_MODEL, "Count to five")
|
||||
require_successful_call(result)
|
||||
assert result.chunks > 0, "streaming passthrough produced no events"
|
||||
|
||||
|
|
@ -113,12 +145,22 @@ def test_gemini_passthrough_streaming_logs_cost(
|
|||
assert row.custom_llm_provider == "gemini"
|
||||
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_MODEL,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_gemini_passthrough_tool_call_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.gemini_generate(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
GEMINI_MODEL,
|
||||
"What is the weather in Paris? Use the get_weather tool.",
|
||||
tools=[
|
||||
GeminiTool(
|
||||
|
|
@ -146,10 +188,19 @@ def test_gemini_passthrough_tool_call_logs_cost(
|
|||
# ---- Anthropic passthrough ---------------------------------------------
|
||||
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_PASSTHROUGH_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_passthrough_nonstreaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.anthropic_message(scoped_key, "claude-haiku-4-5", "Say hello")
|
||||
result = client.anthropic_message(scoped_key, ANTHROPIC_PASSTHROUGH_MODEL, "Say hello")
|
||||
require_successful_call(result)
|
||||
|
||||
row = _fetch_cost_breakdown(client, anthropic_message_id(result))
|
||||
|
|
@ -157,11 +208,20 @@ def test_anthropic_passthrough_nonstreaming_logs_cost(
|
|||
assert "claude" in (row.model or "")
|
||||
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_PASSTHROUGH_MODEL,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_passthrough_streaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.anthropic_message(
|
||||
scoped_key, "claude-haiku-4-5", "Count to five", stream=True
|
||||
scoped_key, ANTHROPIC_PASSTHROUGH_MODEL, "Count to five", stream=True
|
||||
)
|
||||
require_successful_call(result)
|
||||
assert result.chunks > 0, "streaming passthrough produced no events"
|
||||
|
|
@ -170,12 +230,22 @@ def test_anthropic_passthrough_streaming_logs_cost(
|
|||
assert row.custom_llm_provider == "anthropic"
|
||||
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_PASSTHROUGH_MODEL,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_anthropic_passthrough_tool_call_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.anthropic_message(
|
||||
scoped_key,
|
||||
"claude-haiku-4-5",
|
||||
ANTHROPIC_PASSTHROUGH_MODEL,
|
||||
"What is the weather in Paris? Use the get_weather tool.",
|
||||
tools=[
|
||||
AnthropicTool(
|
||||
|
|
@ -205,13 +275,21 @@ class TestPassthroughModelAllowlist:
|
|||
"""
|
||||
|
||||
@pytest.mark.covers("other.auth.passthrough.model_allowlist_enforced")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PROXY_AUTH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(),
|
||||
models=(),
|
||||
)
|
||||
)
|
||||
def test_passthrough_denies_model_outside_key_allowlist(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
key = client.proxy.generate_key(KeyGenerateBody(models=["gemini-2.5-flash"]))
|
||||
key = client.proxy.generate_key(KeyGenerateBody(models=[GEMINI_MODEL]))
|
||||
resources.defer(lambda: client.proxy.delete_key(key))
|
||||
|
||||
result = client.anthropic_message(key, "claude-haiku-4-5", f"say hi {unique_marker()}")
|
||||
result = client.anthropic_message(key, ANTHROPIC_PASSTHROUGH_MODEL, f"say hi {unique_marker()}")
|
||||
assert result.status_code == 403, (
|
||||
"a key restricted to gemini-2.5-flash must be denied a claude passthrough call, "
|
||||
f"got {result.status_code}: {result.body[:300]}"
|
||||
|
|
@ -230,6 +308,14 @@ class TestOpenAIPassthroughPrefix:
|
|||
"""
|
||||
|
||||
@pytest.mark.covers("llm.files.openai.passthrough.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(),
|
||||
)
|
||||
)
|
||||
def test_passthrough_prefix_uploads_a_file_to_openai(
|
||||
self, client: PassthroughClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -252,6 +338,14 @@ class TestOpenAIPassthroughPrefix:
|
|||
assert uploaded.bytes == len(content)
|
||||
|
||||
@pytest.mark.covers("llm.batches.openai.passthrough.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(),
|
||||
)
|
||||
)
|
||||
def test_passthrough_prefix_lists_batches_from_openai(
|
||||
self, client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -274,6 +368,15 @@ class TestOpenAIPassthroughSpend:
|
|||
"""
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.passthrough.stream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(CHEAP_OPENAI_MODEL,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_streamed_responses_call_logs_its_cost(
|
||||
self, client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -318,6 +421,15 @@ class TestOpenAIPassthroughSpend:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.embeddings.openai.passthrough.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(EMBEDDING_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_embeddings_call_logs_its_cost(
|
||||
self, client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -353,6 +465,15 @@ class TestOpenAIProviderPrefixChat:
|
|||
"""
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.openai.passthrough.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(CHEAP_OPENAI_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_prefix_chat_returns_completion_and_logs_its_cost(
|
||||
self, client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -393,6 +514,15 @@ class TestOpenAIPassthroughWebsocket:
|
|||
"""
|
||||
|
||||
@pytest.mark.covers("llm.realtime.openai.passthrough.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(REALTIME_MODEL,),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
)
|
||||
def test_realtime_upgrade_reaches_openai_through_the_passthrough_prefix(
|
||||
self, client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -414,6 +544,15 @@ class TestOpenAIPassthroughWebsocket:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.passthrough_websocket.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(),
|
||||
mode=Mode.WEBSOCKET,
|
||||
)
|
||||
)
|
||||
def test_responses_upgrade_is_accepted_on_the_openai_prefix(
|
||||
self, client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ from pydantic import BaseModel, Field
|
|||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import AuthHeaders, NoBody, require_successful_call, unwrap
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import AnthropicMessagesResponse, ChatMessage, KeyGenerateBody
|
||||
from passthrough_client import PassthroughClient
|
||||
|
|
@ -136,6 +137,15 @@ class TestPassthroughHeaders:
|
|||
"other.config.passthrough.headers_forwarded",
|
||||
exercised_on=[],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.PASSTHROUGH,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_static_and_x_pass_headers_reach_upstream(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -16,10 +16,13 @@ Prompt caching lives in test_cache_control.py.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, LiteLLMParamsBody
|
||||
from passthrough_client import PassthroughClient
|
||||
|
|
@ -27,12 +30,22 @@ from passthrough_client import PassthroughClient
|
|||
pytestmark = pytest.mark.e2e
|
||||
|
||||
SERVICE_TIER = "priority"
|
||||
OPENAI_BACKEND: Final = "openai/gpt-5.5"
|
||||
|
||||
|
||||
class TestServiceTier:
|
||||
@pytest.mark.covers(
|
||||
"llm.chat_completions.openai.service_tier.nonstream.works", exercised_on=[]
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_openai_service_tier_is_echoed(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -40,7 +53,7 @@ class TestServiceTier:
|
|||
model_id = client.proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY"
|
||||
model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY"
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: client.proxy.delete_model(model_id))
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ from __future__ import annotations
|
|||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import NoBody, assert_auth_denied, unwrap
|
||||
from e2e_metadata import Domain, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -59,6 +60,14 @@ def _register(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, str]
|
|||
|
||||
class TestRealtimeHttp:
|
||||
@pytest.mark.covers("llm.realtime.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.REALTIME,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(REALTIME_BACKEND,),
|
||||
)
|
||||
)
|
||||
def test_create_client_secret(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
secret = unwrap(
|
||||
|
|
@ -82,6 +91,7 @@ class TestRealtimeHttp:
|
|||
assert secret.session.type in (None, "realtime"), f"unexpected session type: {secret.session.type}"
|
||||
|
||||
@pytest.mark.covers("other.auth.realtime.missing_header_denied")
|
||||
@meta(Subject(domain=Domain.PROXY_AUTH, route=Route.REALTIME))
|
||||
def test_client_secret_missing_auth_is_denied(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model, _ = _register(proxy, resources)
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -92,6 +102,7 @@ class TestRealtimeHttp:
|
|||
assert_auth_denied(result, "realtime client_secrets missing auth")
|
||||
|
||||
@pytest.mark.covers("other.auth.realtime.missing_header_denied")
|
||||
@meta(Subject(domain=Domain.PROXY_AUTH, route=Route.REALTIME))
|
||||
def test_calls_without_auth_is_denied(self, proxy: ProxyClient) -> None:
|
||||
result = proxy.transport.send(
|
||||
"/v1/realtime/calls",
|
||||
|
|
|
|||
|
|
@ -9,9 +9,12 @@ litellm-regression-tests/tests/test_inference_endpoints.py.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody, RerankBody, RerankResponse
|
||||
from proxy_client import ProxyClient
|
||||
|
|
@ -24,6 +27,8 @@ DOCUMENTS = [
|
|||
"Washington, D.C. is the capital of the United States.",
|
||||
"Capital punishment has existed in the United States since before it was a country.",
|
||||
]
|
||||
COHERE_RERANK_BACKEND: Final = "cohere/rerank-v3.5"
|
||||
BEDROCK_RERANK_BACKEND: Final = "bedrock/arn:aws:bedrock:us-east-1::foundation-model/cohere.rerank-v3-5:0"
|
||||
QUERY = "What is the capital of the United States?"
|
||||
|
||||
|
||||
|
|
@ -43,11 +48,20 @@ def _rerank_top_3(proxy: ProxyClient, key: str, model: str) -> RerankResponse:
|
|||
|
||||
class TestRerank:
|
||||
@pytest.mark.covers("llm.rerank.cohere.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RERANK,
|
||||
providers=(Provider.COHERE,),
|
||||
models=(COHERE_RERANK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_rerank_scores_top_n(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model = f"e2e-rerank-{unique_marker()}"
|
||||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(model="cohere/rerank-v3.5", api_key="os.environ/COHERE_API_KEY"),
|
||||
LiteLLMParamsBody(model=COHERE_RERANK_BACKEND, api_key="os.environ/COHERE_API_KEY"),
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
key = resources.key()
|
||||
|
|
@ -55,6 +69,15 @@ class TestRerank:
|
|||
_assert_top_n_scored(_rerank_top_3(proxy, key, model))
|
||||
|
||||
@pytest.mark.covers("llm.rerank.bedrock.basic.nonstream.works", exercised_on=["rerank"])
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RERANK,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_RERANK_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_bedrock_rerank_scores_top_n(
|
||||
self, proxy: ProxyClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -62,7 +85,7 @@ class TestRerank:
|
|||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="bedrock/arn:aws:bedrock:us-east-1::foundation-model/cohere.rerank-v3-5:0",
|
||||
model=BEDROCK_RERANK_BACKEND,
|
||||
aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID",
|
||||
aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY",
|
||||
aws_region_name="os.environ/AWS_REGION",
|
||||
|
|
|
|||
|
|
@ -23,13 +23,14 @@ from pydantic import BaseModel, Field
|
|||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import StreamingResponse
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatTool, ChatToolFunction, LiteLLMParamsBody
|
||||
from passthrough_client import PassthroughClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
RESPONSES_ONLY_BACKEND = "openai/gpt-5.3-codex"
|
||||
RESPONSES_ONLY_BACKEND: Final = "openai/gpt-5.3-codex"
|
||||
|
||||
|
||||
class _BridgeToolCallFunction(BaseModel):
|
||||
|
|
@ -99,6 +100,15 @@ class TestResponsesBridgeChatCompletionsStreaming:
|
|||
"llm.chat_completions.openai.basic.stream.bridge_shares_chunk_id",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(RESPONSES_ONLY_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_bridged_stream_shares_one_chunk_id(
|
||||
self, client: PassthroughClient, resources: ResourceManager, bridged_model: str
|
||||
) -> None:
|
||||
|
|
@ -124,6 +134,15 @@ class TestResponsesBridgeChatCompletionsStreaming:
|
|||
"llm.chat_completions.openai.basic.stream.bridge_streams_sse",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(RESPONSES_ONLY_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_bridged_stream_delivers_content_finish_reason_and_done(
|
||||
self, client: PassthroughClient, resources: ResourceManager, bridged_model: str
|
||||
) -> None:
|
||||
|
|
@ -149,6 +168,16 @@ class TestResponsesBridgeChatCompletionsStreaming:
|
|||
"llm.chat_completions.openai.tool_use.stream.bridge_streams_tool_call",
|
||||
exercised_on=["chat_completions"],
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(RESPONSES_ONLY_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_bridged_stream_reassembles_tool_call(
|
||||
self, client: PassthroughClient, resources: ResourceManager, bridged_model: str
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -21,6 +21,7 @@ import openai
|
|||
import pytest
|
||||
from e2e_config import PROVIDER_EDGE_ADVERTISE_HOST, PROVIDER_EDGE_BIND_HOST, unique_marker
|
||||
from e2e_http import assert_client_error
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, LiteLLMParamsBody
|
||||
from openai.types.responses import (
|
||||
|
|
@ -52,7 +53,10 @@ class _OptionalResponsesBody(BaseModel):
|
|||
max_output_tokens: int | None = None
|
||||
|
||||
|
||||
BEDROCK_CONVERSE_BACKEND = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"
|
||||
OPENAI_MINI_BACKEND: Final = "openai/gpt-4o-mini"
|
||||
OPENAI_VISION_BACKEND: Final = "openai/gpt-4o"
|
||||
ANTHROPIC_BACKEND: Final = "anthropic/claude-haiku-4-5"
|
||||
BEDROCK_CONVERSE_BACKEND: Final = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"
|
||||
VERTEX_BACKEND: Final = "vertex_ai/gemini-2.5-flash"
|
||||
AZURE_OPENAI_BACKEND: Final = "azure/gpt-5.4-nano"
|
||||
AZURE_OPENAI_API_VERSION: Final = "v1"
|
||||
|
|
@ -100,11 +104,11 @@ WEATHER_TOOL: FunctionToolParam = {
|
|||
|
||||
|
||||
def _openai_params() -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model="openai/gpt-4o-mini", api_key="os.environ/OPENAI_API_KEY")
|
||||
return LiteLLMParamsBody(model=OPENAI_MINI_BACKEND, api_key="os.environ/OPENAI_API_KEY")
|
||||
|
||||
|
||||
def _anthropic_params() -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model="anthropic/claude-haiku-4-5", api_key="os.environ/ANTHROPIC_API_KEY")
|
||||
return LiteLLMParamsBody(model=ANTHROPIC_BACKEND, api_key="os.environ/ANTHROPIC_API_KEY")
|
||||
|
||||
|
||||
def _bedrock_params() -> LiteLLMParamsBody:
|
||||
|
|
@ -160,6 +164,15 @@ class WeatherArguments(BaseModel):
|
|||
|
||||
class TestResponses:
|
||||
@pytest.mark.covers("llm.responses.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MINI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -172,6 +185,15 @@ class TestResponses:
|
|||
assert response.output_text.strip(), f"/responses returned no output text: {response.output!r}"
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.basic.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MINI_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_streaming_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -194,6 +216,15 @@ class TestResponses:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.basic.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MINI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_logs_cost(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None:
|
||||
model = _register(proxy, resources, _openai_params())
|
||||
client = sdk.openai(resources.key())
|
||||
|
|
@ -219,6 +250,16 @@ class TestResponses:
|
|||
assert "gpt-4o-mini" in (row.model or ""), f"unexpected spend row model: {row.model}"
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MINI_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_returns_function_call(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -235,13 +276,23 @@ class TestResponses:
|
|||
_assert_weather_call(response)
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.vision.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_VISION_BACKEND,),
|
||||
capabilities=(Capability.VISION,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_vision_describes_image(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model = _register(
|
||||
proxy,
|
||||
resources,
|
||||
LiteLLMParamsBody(model="openai/gpt-4o", api_key="os.environ/OPENAI_API_KEY"),
|
||||
LiteLLMParamsBody(model=OPENAI_VISION_BACKEND, api_key="os.environ/OPENAI_API_KEY"),
|
||||
)
|
||||
client = sdk.openai(resources.key())
|
||||
|
||||
|
|
@ -264,6 +315,15 @@ class TestResponses:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.responses.anthropic.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_anthropic_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -276,6 +336,16 @@ class TestResponses:
|
|||
assert response.output_text.strip(), f"/responses returned no output text: {response.output!r}"
|
||||
|
||||
@pytest.mark.covers("llm.responses.anthropic.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_anthropic_returns_function_call(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -292,6 +362,15 @@ class TestResponses:
|
|||
_assert_weather_call(response)
|
||||
|
||||
@pytest.mark.covers("llm.responses.bedrock_converse.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_bedrock_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -304,6 +383,16 @@ class TestResponses:
|
|||
assert response.output_text.strip(), f"/responses over bedrock returned no output text: {response.output!r}"
|
||||
|
||||
@pytest.mark.covers("llm.responses.bedrock_converse.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_bedrock_returns_function_call(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -320,6 +409,15 @@ class TestResponses:
|
|||
_assert_weather_call(response)
|
||||
|
||||
@pytest.mark.covers("llm.responses.vertex.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_vertex_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -332,6 +430,16 @@ class TestResponses:
|
|||
assert response.output_text.strip(), f"/responses over vertex returned no output text: {response.output!r}"
|
||||
|
||||
@pytest.mark.covers("llm.responses.vertex.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_vertex_returns_function_call(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -349,6 +457,15 @@ class TestResponses:
|
|||
_assert_weather_call(response)
|
||||
|
||||
@pytest.mark.covers("llm.responses.azure_openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_azure_openai_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -363,6 +480,16 @@ class TestResponses:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.responses.azure_openai.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_OPENAI_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_azure_openai_returns_function_call(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -380,7 +507,37 @@ class TestResponses:
|
|||
_assert_weather_call(response)
|
||||
|
||||
@pytest.mark.provider_edge_host
|
||||
@pytest.mark.parametrize("endpoint", ["/v1/responses", "/v1/chat/completions"])
|
||||
@pytest.mark.parametrize(
|
||||
"endpoint",
|
||||
[
|
||||
pytest.param(
|
||||
"/v1/responses",
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
),
|
||||
id="/v1/responses",
|
||||
),
|
||||
pytest.param(
|
||||
"/v1/chat/completions",
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.BEDROCK,),
|
||||
models=(BEDROCK_CONVERSE_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
),
|
||||
id="/v1/chat/completions",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_bedrock_forwards_allowed_safety_identifier_as_additional_model_request_field(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, endpoint: str
|
||||
) -> None:
|
||||
|
|
@ -442,6 +599,14 @@ class TestResponses:
|
|||
reason="stage red: product gap, /v1/responses 500s (aresponses TypeError) on missing input instead of 400"
|
||||
)
|
||||
@pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MINI_BACKEND,),
|
||||
)
|
||||
)
|
||||
def test_missing_input_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model = _register(proxy, resources, _openai_params(), prefix="e2e-responses-val")
|
||||
key = resources.key()
|
||||
|
|
@ -453,6 +618,12 @@ class TestResponses:
|
|||
assert_client_error(result, "responses missing input")
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
)
|
||||
)
|
||||
def test_missing_model_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
result = proxy.transport.send(
|
||||
|
|
@ -463,6 +634,14 @@ class TestResponses:
|
|||
assert_client_error(result, "responses missing model")
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MINI_BACKEND,),
|
||||
)
|
||||
)
|
||||
def test_empty_input_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model = _register(proxy, resources, _openai_params(), prefix="e2e-responses-val")
|
||||
key = resources.key()
|
||||
|
|
@ -509,6 +688,16 @@ class TodayReport(BaseModel):
|
|||
|
||||
|
||||
class TestResponsesOpenAIHostedFeatures:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(REASONING_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING, Capability.REASONING, Capability.RESPONSE_SCHEMA),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_reasoning_items_replay_into_structured_output_after_tool_call(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -561,6 +750,15 @@ class TestResponsesOpenAIHostedFeatures:
|
|||
assert TOOL_DATE in report.today, f"structured output ignored the tool result: {report!r}"
|
||||
|
||||
@pytest.mark.provider_live
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(SHELL_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_shell_tool_stream_surfaces_shell_call_and_its_output(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ import openai
|
|||
import pytest
|
||||
from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, unique_marker
|
||||
from e2e_http import NoBody, Success, UnknownApiError, unwrap
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from openai.types.responses import (
|
||||
|
|
@ -27,6 +28,7 @@ from sdk_clients import NO_PROXY_CACHE, SdkClients
|
|||
pytestmark = pytest.mark.e2e
|
||||
|
||||
OPENAI_BACKEND: Final = "openai/gpt-5.5"
|
||||
OPENAI_MINI_BACKEND: Final = "openai/gpt-4o-mini"
|
||||
LONG_TASK: Final = "Write a numbered list counting from 1 to 400, one number per line, with a short word after each."
|
||||
CANCELLABLE_STATUSES: Final = frozenset({"queued", "in_progress"})
|
||||
|
||||
|
|
@ -69,11 +71,20 @@ class TestResponsesRetrieve:
|
|||
reason="stage red: product gap (LIT-5446), retrieve returns a different id than the stored response (non-idempotent response-id re-encryption)"
|
||||
)
|
||||
@pytest.mark.covers("llm.responses.openai.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_MINI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_store_and_retrieve_by_id(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
model = f"e2e-resp-store-{unique_marker()}"
|
||||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(model="openai/gpt-4o-mini", api_key="os.environ/OPENAI_API_KEY"),
|
||||
LiteLLMParamsBody(model=OPENAI_MINI_BACKEND, api_key="os.environ/OPENAI_API_KEY"),
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
key = resources.key()
|
||||
|
|
@ -103,6 +114,12 @@ class TestResponsesRetrieve:
|
|||
reason="stage red: product gap (LIT-5447), retrieving an unknown response id returns 400 (model=None) instead of 404"
|
||||
)
|
||||
@pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
)
|
||||
)
|
||||
def test_invalid_response_id_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
get_result = proxy.transport.get(
|
||||
|
|
@ -135,6 +152,15 @@ def _input_texts(item: object) -> tuple[str, ...]:
|
|||
|
||||
@pytest.mark.provider_live
|
||||
class TestStoredResponseLifecycle:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_input_items_list_the_stored_prompt(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -150,6 +176,15 @@ class TestStoredResponseLifecycle:
|
|||
texts = tuple(text for item in items for text in _input_texts(item))
|
||||
assert any(marker in text for text in texts), f"input_items did not list the stored prompt: {items!r}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_deleted_response_is_no_longer_retrievable(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -171,6 +206,15 @@ class TestStoredResponseLifecycle:
|
|||
|
||||
@pytest.mark.provider_live
|
||||
class TestBackgroundResponseCancel:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_cancel_background_response(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -185,6 +229,15 @@ class TestBackgroundResponseCancel:
|
|||
cancelled = client.responses.cancel(created.id)
|
||||
assert cancelled.status == "cancelled", f"cancel did not stop the response: {cancelled.status}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_cancel_background_streaming_response_by_streamed_id(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -15,6 +15,7 @@ from typing import Final, Literal
|
|||
|
||||
import pytest
|
||||
from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS, unique_marker
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody, SpendLogRow
|
||||
from openai import OpenAI
|
||||
|
|
@ -116,6 +117,15 @@ def _assert_spend_row_matches(proxy: ProxyClient, key: str, header_cost: float)
|
|||
|
||||
class TestSailChatCompletions:
|
||||
@pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.SAIL,),
|
||||
models=(BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
("service_tier", "billed_tier"), [("balanced", "balanced"), ("auto", "base")]
|
||||
)
|
||||
|
|
@ -150,6 +160,15 @@ class TestSailChatCompletions:
|
|||
_assert_spend_row_matches(proxy, key, header_cost)
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.SAIL,),
|
||||
models=(BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
@pytest.mark.parametrize("service_tier", ["bogus", 5])
|
||||
def test_unknown_service_tier_is_dropped_and_billed_asap(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, service_tier: str | int
|
||||
|
|
@ -176,6 +195,15 @@ class TestSailChatCompletions:
|
|||
|
||||
class TestSailResponses:
|
||||
@pytest.mark.covers("llm.responses.sail.service_tier.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.SAIL,),
|
||||
models=(BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_caller_completion_window_bills_its_rates(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
@ -204,6 +232,15 @@ class TestSailResponses:
|
|||
|
||||
class TestSailMessages:
|
||||
@pytest.mark.covers("llm.messages.sail.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.SAIL,),
|
||||
models=(BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_plain_call_returns_a_message(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ from typing import Final
|
|||
import pytest
|
||||
from e2e_config import STREAM_MIN_LEAD_SECONDS, provider_paces_stream, unique_marker
|
||||
from e2e_http import StreamingResponse, require_successful_call, unwrap
|
||||
from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import (
|
||||
AnthropicAssistantTurn,
|
||||
|
|
@ -324,6 +325,15 @@ def _weather_call(client: PassthroughClient, key: str, model: str) -> OutMessage
|
|||
|
||||
class TestTogetherChatCompletions:
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.thinking.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_reasoning_surfaces_as_reasoning_content(
|
||||
self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str
|
||||
) -> None:
|
||||
|
|
@ -347,6 +357,15 @@ class TestTogetherChatCompletions:
|
|||
assert message.content and "43" in message.content, f"answer lost: {message}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.thinking.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_reasoning_streams_as_reasoning_content_deltas(
|
||||
self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str
|
||||
) -> None:
|
||||
|
|
@ -369,6 +388,15 @@ class TestTogetherChatCompletions:
|
|||
assert "43" in content, f"streamed answer lost: {content!r}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_tool_call_is_returned(
|
||||
self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str
|
||||
) -> None:
|
||||
|
|
@ -376,6 +404,15 @@ class TestTogetherChatCompletions:
|
|||
_ = _weather_call_ids(_weather_call(client, key, model))
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.tool_use.stream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_tool_call_is_streamed(
|
||||
self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str
|
||||
) -> None:
|
||||
|
|
@ -407,6 +444,15 @@ class TestTogetherChatCompletions:
|
|||
assert "paris" in args.location.lower(), f"streamed tool arguments lost the location: {args}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.multi_turn.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_tool_result_round_trip(
|
||||
self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str
|
||||
) -> None:
|
||||
|
|
@ -443,6 +489,16 @@ class TestTogetherChatCompletions:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.thinking.nonstream.template_kwargs_forwarded")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
models=(HYBRID_REASONING_BACKEND,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_chat_template_kwargs_reach_together(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -476,6 +532,16 @@ class TestTogetherChatCompletions:
|
|||
assert treatment.content and "43" in treatment.content, f"answer lost: {treatment}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.thinking.nonstream.replayed_reasoning_forwarded")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
models=(REASONING_REPLAY_BACKEND,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_replayed_reasoning_content_reaches_together(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -496,6 +562,14 @@ class TestTogetherChatCompletions:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.basic.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_cost_header_and_spend_row_match_the_registry_price(
|
||||
self,
|
||||
client: PassthroughClient,
|
||||
|
|
@ -551,6 +625,16 @@ class TestTogetherChatCompletions:
|
|||
)
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.thinking.nonstream.effort_none_disables")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
models=(HYBRID_REASONING_BACKEND,),
|
||||
capabilities=(Capability.REASONING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_reasoning_effort_none_reaches_together(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -584,6 +668,16 @@ class TestTogetherChatCompletions:
|
|||
assert treatment.content and "43" in treatment.content, f"answer lost: {treatment}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.structured_output.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
models=(HYBRID_REASONING_BACKEND,),
|
||||
capabilities=(Capability.RESPONSE_SCHEMA,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_response_format_json_schema_shapes_the_reply(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -608,6 +702,15 @@ class TestTogetherChatCompletions:
|
|||
assert person.name, f"schema-shaped reply carries an empty name: {message.content!r}"
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.together_ai.prompt_cache_5m.nonstream.cost_logged")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.CHAT_COMPLETIONS,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
capabilities=(Capability.PROMPT_CACHING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_cache_read_tokens_bill_at_the_cache_read_rate(
|
||||
self,
|
||||
client: PassthroughClient,
|
||||
|
|
@ -705,6 +808,15 @@ def _messages_weather_call(
|
|||
|
||||
class TestTogetherMessages:
|
||||
@pytest.mark.covers("llm.messages.together_ai.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_tool_use_block_is_returned(
|
||||
self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str
|
||||
) -> None:
|
||||
|
|
@ -712,6 +824,15 @@ class TestTogetherMessages:
|
|||
_messages_weather_call(client, key, model)
|
||||
|
||||
@pytest.mark.covers("llm.messages.together_ai.multi_turn.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_tool_result_round_trip(
|
||||
self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str
|
||||
) -> None:
|
||||
|
|
@ -744,6 +865,14 @@ class TestTogetherMessages:
|
|||
|
||||
@pytest.mark.covers("llm.messages.together_ai.basic.stream.works")
|
||||
@pytest.mark.provider_live
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.MESSAGES,
|
||||
providers=(Provider.TOGETHER_AI,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_streams_text_deltas(
|
||||
self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -9,15 +9,43 @@ route the claude_code rows never reach
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import require_successful_call
|
||||
from e2e_metadata import Domain, Provider, Route, Subject, meta
|
||||
from proxy_client import ProxyClient
|
||||
from pydantic import BaseModel
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
GEMINI_DEPLOYMENTS = ("gemini-2.5-flash", "gemini-2.5-flash-vertex")
|
||||
GEMINI_STUDIO_DEPLOYMENT: Final = "gemini-2.5-flash"
|
||||
GEMINI_VERTEX_DEPLOYMENT: Final = "gemini-2.5-flash-vertex"
|
||||
GEMINI_DEPLOYMENTS = (
|
||||
pytest.param(
|
||||
GEMINI_STUDIO_DEPLOYMENT,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.COUNT_TOKENS,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_STUDIO_DEPLOYMENT,),
|
||||
)
|
||||
),
|
||||
),
|
||||
pytest.param(
|
||||
GEMINI_VERTEX_DEPLOYMENT,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.COUNT_TOKENS,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(GEMINI_VERTEX_DEPLOYMENT,),
|
||||
)
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class _Part(BaseModel):
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from typing import Literal
|
|||
|
||||
import pytest
|
||||
from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, unique_marker
|
||||
from e2e_metadata import Domain, Provider, Route, Subject, meta
|
||||
from e2e_http import (
|
||||
FileUploadForm,
|
||||
NoBody,
|
||||
|
|
@ -162,6 +163,7 @@ def _await_store_in_list(proxy: ProxyClient, key: str, store_id: str) -> None:
|
|||
|
||||
class TestVectorStores:
|
||||
@pytest.mark.covers("llm.vector_stores.openai.basic.nonstream.works")
|
||||
@meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,)))
|
||||
def test_create_list_retrieve_delete_lifecycle(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
name = f"e2e-vector-store-{unique_marker()}"
|
||||
|
|
@ -204,6 +206,7 @@ class TestVectorStores:
|
|||
reason="stage red: product gap, vector store search 500s (asearch TypeError) on missing query instead of 400"
|
||||
)
|
||||
@pytest.mark.covers("llm.vector_stores.openai.input_validation.nonstream.works")
|
||||
@meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,)))
|
||||
def test_search_missing_query_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
created = unwrap(
|
||||
|
|
@ -223,6 +226,7 @@ class TestVectorStores:
|
|||
assert_client_error(result, "vector store search missing query")
|
||||
|
||||
@pytest.mark.covers("llm.vector_stores.openai.basic.nonstream.works")
|
||||
@meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,)))
|
||||
def test_file_attach_poll_and_search(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
marker = f"azure-falcon-{unique_marker()}"
|
||||
|
|
@ -309,6 +313,7 @@ class TestVectorStores:
|
|||
reason="stage red: product gap, retrieving a nonexistent vector store returns 2xx with an error envelope in the body instead of 404"
|
||||
)
|
||||
@pytest.mark.covers("llm.vector_stores.openai.input_validation.nonstream.works")
|
||||
@meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,)))
|
||||
def test_retrieve_invalid_id_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
result = proxy.transport.get(
|
||||
|
|
@ -328,6 +333,7 @@ class TestVectorStores:
|
|||
pytest.fail(f"invalid vector store id must be a client error, got {other!r}")
|
||||
|
||||
@pytest.mark.covers("llm.vector_stores.openai.input_validation.nonstream.works")
|
||||
@meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,)))
|
||||
def test_invalid_chunking_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None:
|
||||
key = resources.key()
|
||||
result = proxy.transport.send(
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ from pydantic import BaseModel
|
|||
|
||||
from e2e_config import settle_propagation, unique_marker
|
||||
from e2e_http import NoBody, require_successful_call, unwrap
|
||||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import SpendLogRow
|
||||
from passthrough_client import PassthroughClient
|
||||
|
|
@ -149,6 +150,15 @@ def _costed_row(client: PassthroughClient, call_id: str | None) -> SpendLogRow:
|
|||
|
||||
|
||||
class TestVertexPassthroughSpendTracking:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.SPEND_BUDGETS,
|
||||
route=Route.PASSTHROUGH,
|
||||
providers=(Provider.VERTEX_AI,),
|
||||
models=(VERTEX_MODEL,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_vertex_passthrough_via_managed_model_logs_cost(
|
||||
self,
|
||||
client: PassthroughClient,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue