mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
test(e2e): cover 12 non-core LLM coverage registry cells (#34123)
* fix(e2e): reference client.proxy in mid-conversation native providers test
EndpointsClient exposes the shared ProxyClient as .proxy and has never had a
.gateway attribute, so these two calls raised AttributeError at runtime and
failed the tests/e2e basedpyright zero-error gate for any PR touching e2e
files. Introduced in 23b5b7d199.
* test(e2e): cover 12 non-core LLM coverage registry cells
Raises Non-Core LLMs registry coverage from 24/50 to 36/50 (overall 51.9%
to 54.8%). Four cells were already asserted by existing tests and only
gain their covers marker (openai embeddings, openai image generation,
openai TTS, cohere rerank); one is dual-marked onto the existing
spend-tracking embeddings test rather than duplicated.
New tests: bedrock and vertex embeddings, streaming TTS (asserts chunked
transfer encoding so a buffered body cannot pass), audio transcriptions
via the realtime suite's wav fixture, moderations flag/pass pair, and
files list/retrieve in the batches suite.
Harness: e2e_http.upload generalized to any form model with a
file_content_type override (batches path unchanged), new stream_binary
primitive + BinaryStream for binary chunked responses, transcribe and
moderations client methods, file retrieve/list client methods.
* fix(e2e): close streamed TTS response on error paths and surface the error body
With stream=True a non-2xx response returned with the body unread, keeping
the socket checked out until garbage collection; the sibling
_streaming_outcome already consumes resp.text on error. The response now
closes on every path and BinaryStream carries a bounded error_body so a
failed stream call is triageable.
* test(e2e): assert streamed TTS response carries no content-length
This commit is contained in:
parent
e7b9357bc2
commit
e967bc8c4f
12 changed files with 496 additions and 20 deletions
|
|
@ -26,16 +26,24 @@ from e2e_http import (
|
|||
)
|
||||
from models import LiteLLMParamsBody
|
||||
|
||||
UPLOAD_FILENAME = "batch_input.jsonl"
|
||||
|
||||
|
||||
class FileObject(BaseModel):
|
||||
id: str
|
||||
object: str | None = None
|
||||
purpose: str | None = None
|
||||
filename: str | None = None
|
||||
bytes: int | None = None
|
||||
status: str | None = None
|
||||
created_at: int | None = None
|
||||
|
||||
|
||||
class FileList(BaseModel):
|
||||
object: str | None = None
|
||||
data: list[FileObject] = []
|
||||
|
||||
|
||||
class BatchObject(BaseModel):
|
||||
id: str
|
||||
object: str | None = None
|
||||
|
|
@ -106,12 +114,30 @@ class BatchClient:
|
|||
_files_path(provider),
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
form=form,
|
||||
filename="batch_input.jsonl",
|
||||
filename=UPLOAD_FILENAME,
|
||||
content=content,
|
||||
params=ModelQuery(model=model),
|
||||
response_type=FileObject,
|
||||
)
|
||||
|
||||
def retrieve_file(
|
||||
self, file_id: str, *, key: str, provider: str | None = None
|
||||
) -> Result[FileObject]:
|
||||
return self.proxy.transport.get(
|
||||
f"{_files_path(provider)}/{file_id}",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
params=NoBody(),
|
||||
response_type=FileObject,
|
||||
)
|
||||
|
||||
def list_files(self, *, key: str, provider: str | None = None) -> Result[FileList]:
|
||||
return self.proxy.transport.get(
|
||||
_files_path(provider),
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
params=NoBody(),
|
||||
response_type=FileList,
|
||||
)
|
||||
|
||||
def create_batch(
|
||||
self, *, body: BatchCreateBody, key: str, provider: str | None = None
|
||||
) -> StreamingResponse:
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ import pytest
|
|||
from e2e_config import require_env, unique_marker
|
||||
|
||||
from batch_client import (
|
||||
UPLOAD_FILENAME,
|
||||
BatchClient,
|
||||
BatchCreateBody,
|
||||
BatchObject,
|
||||
|
|
@ -511,6 +512,72 @@ class TestBatchFileContent:
|
|||
)
|
||||
|
||||
|
||||
class TestOpenAIFiles:
|
||||
"""GET /v1/files (list) and GET /v1/files/{id} (retrieve) over the OpenAI route.
|
||||
|
||||
The proxy lists the OpenAI org's raw file ids, so the list case uploads a raw
|
||||
(provider-routed) file whose id matches what list returns; retrieve re-encodes
|
||||
the id it was called with, so the model-encoded upload round-trips unchanged.
|
||||
"""
|
||||
|
||||
@pytest.mark.covers(
|
||||
"llm.files.openai.list.nonstream.works",
|
||||
exercised_on=["files"],
|
||||
)
|
||||
def test_uploaded_file_appears_in_list(
|
||||
self, client: BatchClient, resources: ResourceManager, batch_deployments: None
|
||||
) -> None:
|
||||
key = resources.key()
|
||||
file = unwrap(
|
||||
client.upload_file(
|
||||
content=render_jsonl(OPENAI_BATCH_MODEL),
|
||||
form=FileUploadForm(purpose="batch"),
|
||||
key=key,
|
||||
provider="openai",
|
||||
)
|
||||
)
|
||||
resources.defer(
|
||||
quietly(lambda: client.delete_file(file.id, key=key, provider="openai"))
|
||||
)
|
||||
|
||||
listed = unwrap(client.list_files(key=key))
|
||||
assert listed.object is None or listed.object == "list", (
|
||||
f"list envelope object={listed.object!r}"
|
||||
)
|
||||
match = next((entry for entry in listed.data if entry.id == file.id), None)
|
||||
assert match is not None, f"uploaded file {file.id!r} absent from GET /v1/files"
|
||||
assert match.purpose == "batch", (
|
||||
f"listed file must round-trip the upload purpose, got {match.purpose!r}"
|
||||
)
|
||||
|
||||
@pytest.mark.covers(
|
||||
"llm.files.openai.retrieve.nonstream.works",
|
||||
exercised_on=["files"],
|
||||
)
|
||||
def test_retrieve_round_trips_metadata(
|
||||
self, client: BatchClient, resources: ResourceManager, batch_deployments: None
|
||||
) -> None:
|
||||
key = resources.key()
|
||||
file = unwrap(
|
||||
client.upload_file(
|
||||
content=render_jsonl(OPENAI_BATCH_MODEL),
|
||||
form=FileUploadForm(purpose="batch"),
|
||||
model=OPENAI_BATCH_MODEL,
|
||||
key=key,
|
||||
)
|
||||
)
|
||||
resources.defer(quietly(lambda: client.delete_file(file.id, key=key)))
|
||||
|
||||
fetched = unwrap(client.retrieve_file(file.id, key=key))
|
||||
assert fetched.id == file.id, "retrieve must echo the uploaded file id"
|
||||
assert fetched.purpose == "batch", (
|
||||
f"retrieve must round-trip purpose, got {fetched.purpose!r}"
|
||||
)
|
||||
assert fetched.filename == UPLOAD_FILENAME, (
|
||||
f"retrieve must round-trip filename, got {fetched.filename!r}"
|
||||
)
|
||||
|
||||
|
||||
BATCH_RL_REQUEST_LINES = 3
|
||||
BATCH_RL_RPM_LIMIT = 2
|
||||
|
||||
|
|
|
|||
|
|
@ -147,6 +147,32 @@ class StreamingResponse(BaseModel):
|
|||
return "text/event-stream" in (self.content_type or "")
|
||||
|
||||
|
||||
class BinaryStream(BaseModel):
|
||||
"""Outcome of consuming a binary chunked response (e.g. TTS audio) as a stream.
|
||||
|
||||
Unlike StreamingResponse, which line-splits an SSE text body, this iterates the
|
||||
raw bytes with iter_content and reports how many non-empty chunks arrived and
|
||||
the total byte count, so a caller can assert customer-observable streaming
|
||||
(multiple chunks, real bytes) without decoding the payload."""
|
||||
|
||||
status_code: int
|
||||
content_type: str | None = None
|
||||
call_id: str | None = None
|
||||
transfer_encoding: str | None = None
|
||||
content_length: str | None = None
|
||||
error_body: str | None = None
|
||||
chunk_count: int = 0
|
||||
total_bytes: int = 0
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
return 200 <= self.status_code < 300
|
||||
|
||||
@property
|
||||
def chunked(self) -> bool:
|
||||
return "chunked" in (self.transfer_encoding or "")
|
||||
|
||||
|
||||
def _hdr(resp: requests.Response, name: str) -> str | None:
|
||||
value = resp.headers.get(name)
|
||||
return value if isinstance(value, str) else None
|
||||
|
|
@ -430,16 +456,18 @@ def upload[R: BaseModel](
|
|||
url: URL,
|
||||
*,
|
||||
headers: BaseModel,
|
||||
form: FileUploadForm,
|
||||
form: BaseModel,
|
||||
filename: str,
|
||||
content: bytes,
|
||||
file_content_type: str = "application/jsonl",
|
||||
params: BaseModel | None = None,
|
||||
response_type: type[R],
|
||||
timeout: float = 60.0,
|
||||
) -> Result[R]:
|
||||
"""Multipart POST for file uploads (/v1/files). Form fields come from `form`,
|
||||
the file bytes are sent as the `file` part, and `params` carries any query
|
||||
routing (e.g. ?model=). requests sets the multipart Content-Type itself."""
|
||||
"""Multipart POST for file-bearing routes (/v1/files, /v1/audio/transcriptions).
|
||||
Form fields come from `form`, the file bytes are sent as the `file` part with
|
||||
`file_content_type`, and `params` carries any query routing (e.g. ?model=).
|
||||
requests sets the multipart Content-Type itself."""
|
||||
dumped: dict[str, object] = form.model_dump(by_alias=True, exclude_none=True)
|
||||
data = {key: str(value) for key, value in dumped.items()}
|
||||
try:
|
||||
|
|
@ -448,7 +476,7 @@ def upload[R: BaseModel](
|
|||
headers=_headers(headers),
|
||||
params=_params(params),
|
||||
data=data,
|
||||
files={"file": (filename, content, "application/jsonl")},
|
||||
files={"file": (filename, content, file_content_type)},
|
||||
timeout=timeout,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
|
|
@ -456,6 +484,54 @@ def upload[R: BaseModel](
|
|||
return _classify(resp, response_type)
|
||||
|
||||
|
||||
def stream_binary(
|
||||
url: URL,
|
||||
*,
|
||||
headers: BaseModel,
|
||||
json: BaseModel,
|
||||
chunk_size: int = 8192,
|
||||
timeout: float = 60.0,
|
||||
) -> BinaryStream:
|
||||
"""POST that consumes a binary chunked response (e.g. TTS audio) as a stream,
|
||||
counting non-empty chunks and total bytes with iter_content. A non-2xx status
|
||||
short-circuits with the counts left at zero so the caller can fail loudly."""
|
||||
try:
|
||||
resp = requests.post(
|
||||
str(url),
|
||||
headers=_headers(headers),
|
||||
json=json.model_dump(by_alias=True, exclude_none=True),
|
||||
stream=True,
|
||||
timeout=timeout,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return BinaryStream(status_code=-1, error_body=str(exc)[:300])
|
||||
with resp:
|
||||
content_type = _hdr(resp, "content-type")
|
||||
call_id = _hdr(resp, "x-litellm-call-id")
|
||||
transfer_encoding = _hdr(resp, "transfer-encoding")
|
||||
content_length = _hdr(resp, "content-length")
|
||||
if not (200 <= resp.status_code < 300):
|
||||
return BinaryStream(
|
||||
status_code=resp.status_code,
|
||||
content_type=content_type,
|
||||
call_id=call_id,
|
||||
transfer_encoding=transfer_encoding,
|
||||
content_length=content_length,
|
||||
error_body=resp.text[:300],
|
||||
)
|
||||
raw_chunks = cast("Iterator[bytes]", resp.iter_content(chunk_size=chunk_size))
|
||||
chunks = tuple(chunk for chunk in raw_chunks if chunk)
|
||||
return BinaryStream(
|
||||
status_code=resp.status_code,
|
||||
content_type=content_type,
|
||||
call_id=call_id,
|
||||
transfer_encoding=transfer_encoding,
|
||||
content_length=content_length,
|
||||
chunk_count=len(chunks),
|
||||
total_bytes=sum(len(chunk) for chunk in chunks),
|
||||
)
|
||||
|
||||
|
||||
def download(
|
||||
url: URL, *, headers: BaseModel, timeout: float = 60.0
|
||||
) -> StreamingResponse:
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ from typing import Literal
|
|||
from pydantic import BaseModel
|
||||
|
||||
from proxy_client import ProxyClient
|
||||
from e2e_http import StreamingResponse
|
||||
from e2e_http import BinaryStream, Result, StreamingResponse
|
||||
from models import CacheControl, ChatMessage, LiteLLMParamsBody, RichMessage, TextBlock
|
||||
|
||||
__all__ = [
|
||||
|
|
@ -110,6 +110,16 @@ class ImageRequest(BaseModel):
|
|||
size: str = "1024x1024"
|
||||
|
||||
|
||||
class TranscriptionForm(BaseModel):
|
||||
model: str
|
||||
response_format: str = "json"
|
||||
|
||||
|
||||
class ModerationRequest(BaseModel):
|
||||
model: str
|
||||
input: str
|
||||
|
||||
|
||||
class ResponsesOutputContent(BaseModel):
|
||||
type: str | None = None
|
||||
text: str | None = None
|
||||
|
|
@ -213,6 +223,27 @@ class ImagesResult(BaseModel):
|
|||
data: list[ImageItem] = []
|
||||
|
||||
|
||||
class TranscriptionResult(BaseModel):
|
||||
text: str = ""
|
||||
|
||||
|
||||
class ModerationResultItem(BaseModel):
|
||||
flagged: bool
|
||||
categories: dict[str, bool] = {}
|
||||
|
||||
@property
|
||||
def flagged_categories(self) -> tuple[str, ...]:
|
||||
return tuple(name for name, hit in self.categories.items() if hit)
|
||||
|
||||
|
||||
class ModerationResult(BaseModel):
|
||||
results: list[ModerationResultItem] = []
|
||||
|
||||
@property
|
||||
def first(self) -> ModerationResultItem | None:
|
||||
return self.results[0] if self.results else None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class EndpointsClient:
|
||||
proxy: ProxyClient
|
||||
|
|
@ -314,6 +345,36 @@ class EndpointsClient:
|
|||
"/v1/audio/speech", key, SpeechRequest(model=model, input=text, voice=voice)
|
||||
)
|
||||
|
||||
def audio_speech_stream(
|
||||
self, key: str, model: str, text: str, *, voice: str = "alloy"
|
||||
) -> BinaryStream:
|
||||
return self.proxy.transport.stream_binary(
|
||||
"/v1/audio/speech",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
json=SpeechRequest(model=model, input=text, voice=voice),
|
||||
)
|
||||
|
||||
def transcribe(
|
||||
self, key: str, model: str, *, filename: str, content: bytes
|
||||
) -> Result[TranscriptionResult]:
|
||||
return self.proxy.transport.upload(
|
||||
"/v1/audio/transcriptions",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
form=TranscriptionForm(model=model),
|
||||
filename=filename,
|
||||
content=content,
|
||||
file_content_type="audio/wav",
|
||||
response_type=TranscriptionResult,
|
||||
)
|
||||
|
||||
def moderations(self, key: str, model: str, text: str) -> Result[ModerationResult]:
|
||||
return self.proxy.transport.post(
|
||||
"/v1/moderations",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
json=ModerationRequest(model=model, input=text),
|
||||
response_type=ModerationResult,
|
||||
)
|
||||
|
||||
def images(self, key: str, model: str, prompt: str) -> StreamingResponse:
|
||||
return self._send(
|
||||
"/v1/images/generations", key, ImageRequest(model=model, prompt=prompt)
|
||||
|
|
|
|||
|
|
@ -1,8 +1,9 @@
|
|||
"""Live e2e: POST /v1/audio/speech returns audio.
|
||||
"""Live e2e: POST /v1/audio/speech returns audio, non-streamed and streamed.
|
||||
|
||||
Registers an OpenAI text-to-speech deployment at runtime and asserts the response
|
||||
is an audio body (binary, not JSON). Migrated from
|
||||
litellm-regression-tests/tests/test_inference_endpoints.py.
|
||||
The non-streamed call asserts an audio (not JSON) body. The streamed call consumes
|
||||
the response the way a player would and asserts customer-observable streaming:
|
||||
chunked transfer encoding (a buffered body would carry a content-length) with
|
||||
non-zero audio bytes.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -19,6 +20,7 @@ pytestmark = pytest.mark.e2e
|
|||
|
||||
|
||||
class TestAudioSpeech:
|
||||
@pytest.mark.covers("llm.audio_speech.openai.basic.nonstream.works")
|
||||
def test_audio_speech_returns_audio(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -38,3 +40,39 @@ class TestAudioSpeech:
|
|||
f"/audio/speech content-type is not audio: {result.content_type!r}"
|
||||
)
|
||||
assert result.body, "/audio/speech returned an empty body"
|
||||
|
||||
@pytest.mark.covers("llm.audio_speech.openai.basic.stream.works")
|
||||
def test_audio_speech_streams_audio_chunks(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = f"e2e-speech-stream-{unique_marker()}"
|
||||
model_id = endpoints_client.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/gpt-4o-mini-tts", api_key="os.environ/OPENAI_API_KEY"
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: endpoints_client.delete_model(model_id))
|
||||
key = resources.key()
|
||||
|
||||
result = endpoints_client.audio_speech_stream(
|
||||
key,
|
||||
model,
|
||||
"Streaming speech should arrive in several audio chunks so a client can "
|
||||
"begin playback well before the whole clip has finished generating.",
|
||||
)
|
||||
assert result.ok, (
|
||||
f"/audio/speech stream failed (status {result.status_code}); body={result.error_body}"
|
||||
)
|
||||
assert "audio" in (result.content_type or ""), (
|
||||
f"/audio/speech content-type is not audio: {result.content_type!r}"
|
||||
)
|
||||
assert result.chunked, (
|
||||
f"/audio/speech did not stream: transfer-encoding={result.transfer_encoding!r}, "
|
||||
f"content-length={result.content_length!r} (a buffered body is not a stream)"
|
||||
)
|
||||
assert result.content_length is None, (
|
||||
f"/audio/speech advertised content-length={result.content_length!r} on a "
|
||||
f"streamed response (a buffered body is not a stream)"
|
||||
)
|
||||
assert result.total_bytes > 0, "/audio/speech stream returned no audio bytes"
|
||||
|
|
|
|||
51
tests/e2e/llm_translation/test_audio_transcriptions_e2e.py
Normal file
51
tests/e2e/llm_translation/test_audio_transcriptions_e2e.py
Normal file
|
|
@ -0,0 +1,51 @@
|
|||
"""Live e2e: POST /v1/audio/transcriptions turns speech into text.
|
||||
|
||||
Registers an OpenAI speech-to-text deployment at runtime and uploads a spoken
|
||||
weather question (the realtime suite's 24kHz WAV fixture) as multipart, asserting
|
||||
the returned transcript is non-empty and mentions the word it was asked about.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from endpoints_client import EndpointsClient
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
WEATHER_WAV = (
|
||||
Path(__file__).resolve().parent / "realtime" / "fixtures" / "weather_question_24k.wav"
|
||||
)
|
||||
|
||||
|
||||
class TestAudioTranscriptions:
|
||||
@pytest.mark.covers("llm.audio_transcriptions.openai.basic.nonstream.works")
|
||||
def test_audio_transcriptions_returns_text(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = f"e2e-transcribe-{unique_marker()}"
|
||||
model_id = endpoints_client.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/gpt-4o-mini-transcribe", api_key="os.environ/OPENAI_API_KEY"
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: endpoints_client.delete_model(model_id))
|
||||
key = resources.key()
|
||||
|
||||
result = unwrap(
|
||||
endpoints_client.transcribe(
|
||||
key, model, filename=WEATHER_WAV.name, content=WEATHER_WAV.read_bytes()
|
||||
)
|
||||
)
|
||||
text = result.text.strip()
|
||||
assert text, "/audio/transcriptions returned an empty transcript"
|
||||
assert "weather" in text.lower(), (
|
||||
f"transcript of a spoken weather question does not mention weather: {text!r}"
|
||||
)
|
||||
|
|
@ -1,9 +1,9 @@
|
|||
"""Live e2e: POST /embeddings returns a real vector.
|
||||
"""Live e2e: POST /embeddings returns a real vector across OpenAI, Bedrock, Vertex.
|
||||
|
||||
Registers an OpenAI embedding deployment at runtime and asserts a non-empty,
|
||||
non-zero vector came back. Migrated from
|
||||
litellm-regression-tests/tests/test_inference_endpoints.py; the LIT-3167 guard in
|
||||
tests/e2e/embeddings/ covers the Gemini embedding path.
|
||||
Each test registers the deployment it needs at runtime (deleted on teardown) and
|
||||
asserts a non-empty, non-zero vector came back. The LIT-3167 guard in
|
||||
tests/e2e/embeddings/ covers the Gemini embedding path; embeddings cost tracking is
|
||||
covered by tests/e2e/quota_management/spend_tracking/.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -20,6 +20,7 @@ pytestmark = pytest.mark.e2e
|
|||
|
||||
|
||||
class TestEmbeddingsEndpoint:
|
||||
@pytest.mark.covers("llm.embeddings.openai.basic.nonstream.works")
|
||||
def test_embeddings_returns_vector(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -40,3 +41,49 @@ class TestEmbeddingsEndpoint:
|
|||
assert any(component != 0.0 for component in parsed.first_vector), (
|
||||
f"embedding vector is all zeros: {result.body[:300]}"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("llm.embeddings.bedrock.basic.nonstream.works")
|
||||
def test_bedrock_embeddings_returns_vector(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = f"e2e-embeddings-bedrock-{unique_marker()}"
|
||||
model_id = endpoints_client.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="bedrock/amazon.titan-embed-text-v2:0", aws_region_name="us-west-2"
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: endpoints_client.delete_model(model_id))
|
||||
key = resources.key()
|
||||
|
||||
result = endpoints_client.embeddings(key, model, "Say this is a test!")
|
||||
require_successful_call(result)
|
||||
parsed = EmbeddingsResult.model_validate_json(result.body)
|
||||
assert parsed.first_vector, f"/embeddings returned no vector: {result.body[:300]}"
|
||||
assert any(component != 0.0 for component in parsed.first_vector), (
|
||||
f"embedding vector is all zeros: {result.body[:300]}"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("llm.embeddings.vertex.basic.nonstream.works")
|
||||
def test_vertex_embeddings_returns_vector(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = f"e2e-embeddings-vertex-{unique_marker()}"
|
||||
model_id = endpoints_client.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="vertex_ai/gemini-embedding-2",
|
||||
vertex_project="os.environ/VERTEXAI_PROJECT",
|
||||
vertex_location="us-central1",
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: endpoints_client.delete_model(model_id))
|
||||
key = resources.key()
|
||||
|
||||
result = endpoints_client.embeddings(key, model, "Say this is a test!")
|
||||
require_successful_call(result)
|
||||
parsed = EmbeddingsResult.model_validate_json(result.body)
|
||||
assert parsed.first_vector, f"/embeddings returned no vector: {result.body[:300]}"
|
||||
assert any(component != 0.0 for component in parsed.first_vector), (
|
||||
f"embedding vector is all zeros: {result.body[:300]}"
|
||||
)
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ pytestmark = pytest.mark.e2e
|
|||
|
||||
|
||||
class TestImageGeneration:
|
||||
@pytest.mark.covers("llm.images_generations.openai.basic.nonstream.works")
|
||||
def test_image_generation_returns_image(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
65
tests/e2e/llm_translation/test_moderations_e2e.py
Normal file
65
tests/e2e/llm_translation/test_moderations_e2e.py
Normal file
|
|
@ -0,0 +1,65 @@
|
|||
"""Live e2e: POST /v1/moderations classifies content against the provider policy.
|
||||
|
||||
Registers OpenAI's omni moderation model at runtime and asserts the product
|
||||
promise on both sides of the decision: clearly violent text comes back flagged
|
||||
with at least one policy category tripped, and benign text comes back not flagged.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from endpoints_client import EndpointsClient
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
VIOLENT_TEXT = "I am going to find you and kill you, and I will hurt everyone you love."
|
||||
BENIGN_TEXT = "I enjoyed the sunny afternoon and a relaxing walk in the park today."
|
||||
|
||||
|
||||
def _register_moderation_model(
|
||||
endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> str:
|
||||
model = f"e2e-moderation-{unique_marker()}"
|
||||
model_id = endpoints_client.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="openai/omni-moderation-latest", api_key="os.environ/OPENAI_API_KEY"
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: endpoints_client.delete_model(model_id))
|
||||
return model
|
||||
|
||||
|
||||
class TestModerations:
|
||||
@pytest.mark.covers("llm.moderations.openai.basic.nonstream.works")
|
||||
def test_moderations_flags_violent_content(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = _register_moderation_model(endpoints_client, resources)
|
||||
key = resources.key()
|
||||
|
||||
result = unwrap(endpoints_client.moderations(key, model, VIOLENT_TEXT))
|
||||
item = result.first
|
||||
assert item is not None, f"/moderations returned no results: {result}"
|
||||
assert item.flagged, f"violent text was not flagged: {item}"
|
||||
assert item.flagged_categories, (
|
||||
f"flagged result reported no true category: {item}"
|
||||
)
|
||||
|
||||
def test_moderations_passes_benign_content(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = _register_moderation_model(endpoints_client, resources)
|
||||
key = resources.key()
|
||||
|
||||
result = unwrap(endpoints_client.moderations(key, model, BENIGN_TEXT))
|
||||
item = result.first
|
||||
assert item is not None, f"/moderations returned no results: {result}"
|
||||
assert not item.flagged, (
|
||||
f"benign text was flagged as {item.flagged_categories}: {item}"
|
||||
)
|
||||
|
|
@ -26,6 +26,7 @@ DOCUMENTS = [
|
|||
|
||||
|
||||
class TestRerank:
|
||||
@pytest.mark.covers("llm.rerank.cohere.basic.nonstream.works")
|
||||
def test_rerank_scores_top_n(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -194,6 +194,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend(
|
|||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.embeddings.logs_cost")
|
||||
@pytest.mark.covers("llm.embeddings.openai.basic.nonstream.cost_logged")
|
||||
def test_embedding_writes_nonzero_spend_row(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ import e2e_http
|
|||
from e2e_http import (
|
||||
URL,
|
||||
AuthHeaders,
|
||||
FileUploadForm,
|
||||
BinaryStream,
|
||||
ProbeResult,
|
||||
Result,
|
||||
StreamingResponse,
|
||||
|
|
@ -32,6 +32,15 @@ class Transport(Protocol):
|
|||
self, path: str, *, headers: BaseModel, json: BaseModel
|
||||
) -> StreamingResponse: ...
|
||||
|
||||
def stream_binary(
|
||||
self,
|
||||
path: str,
|
||||
*,
|
||||
headers: BaseModel,
|
||||
json: BaseModel,
|
||||
chunk_size: int = 8192,
|
||||
) -> BinaryStream: ...
|
||||
|
||||
def send(
|
||||
self,
|
||||
path: str,
|
||||
|
|
@ -76,9 +85,10 @@ class Transport(Protocol):
|
|||
path: str,
|
||||
*,
|
||||
headers: BaseModel,
|
||||
form: FileUploadForm,
|
||||
form: BaseModel,
|
||||
filename: str,
|
||||
content: bytes,
|
||||
file_content_type: str = "application/jsonl",
|
||||
params: BaseModel | None = None,
|
||||
response_type: type[R],
|
||||
) -> Result[R]: ...
|
||||
|
|
@ -181,6 +191,22 @@ class HttpTransport:
|
|||
self._url(path), headers=headers, json=json, timeout=self.request_timeout
|
||||
)
|
||||
|
||||
def stream_binary(
|
||||
self,
|
||||
path: str,
|
||||
*,
|
||||
headers: BaseModel,
|
||||
json: BaseModel,
|
||||
chunk_size: int = 8192,
|
||||
) -> BinaryStream:
|
||||
return e2e_http.stream_binary(
|
||||
self._url(path),
|
||||
headers=headers,
|
||||
json=json,
|
||||
chunk_size=chunk_size,
|
||||
timeout=self.request_timeout,
|
||||
)
|
||||
|
||||
def send(
|
||||
self,
|
||||
path: str,
|
||||
|
|
@ -212,9 +238,10 @@ class HttpTransport:
|
|||
path: str,
|
||||
*,
|
||||
headers: BaseModel,
|
||||
form: FileUploadForm,
|
||||
form: BaseModel,
|
||||
filename: str,
|
||||
content: bytes,
|
||||
file_content_type: str = "application/jsonl",
|
||||
params: BaseModel | None = None,
|
||||
response_type: type[R],
|
||||
) -> Result[R]:
|
||||
|
|
@ -224,6 +251,7 @@ class HttpTransport:
|
|||
form=form,
|
||||
filename=filename,
|
||||
content=content,
|
||||
file_content_type=file_content_type,
|
||||
params=params,
|
||||
response_type=response_type,
|
||||
timeout=self.request_timeout,
|
||||
|
|
@ -346,6 +374,18 @@ class SplitTransport:
|
|||
) -> StreamingResponse:
|
||||
return self._route(path).stream(path, headers=headers, json=json)
|
||||
|
||||
def stream_binary(
|
||||
self,
|
||||
path: str,
|
||||
*,
|
||||
headers: BaseModel,
|
||||
json: BaseModel,
|
||||
chunk_size: int = 8192,
|
||||
) -> BinaryStream:
|
||||
return self._route(path).stream_binary(
|
||||
path, headers=headers, json=json, chunk_size=chunk_size
|
||||
)
|
||||
|
||||
def send(
|
||||
self,
|
||||
path: str,
|
||||
|
|
@ -367,9 +407,10 @@ class SplitTransport:
|
|||
path: str,
|
||||
*,
|
||||
headers: BaseModel,
|
||||
form: FileUploadForm,
|
||||
form: BaseModel,
|
||||
filename: str,
|
||||
content: bytes,
|
||||
file_content_type: str = "application/jsonl",
|
||||
params: BaseModel | None = None,
|
||||
response_type: type[R],
|
||||
) -> Result[R]:
|
||||
|
|
@ -379,6 +420,7 @@ class SplitTransport:
|
|||
form=form,
|
||||
filename=filename,
|
||||
content=content,
|
||||
file_content_type=file_content_type,
|
||||
params=params,
|
||||
response_type=response_type,
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue