test(e2e): cover 12 non-core LLM coverage registry cells (#34123)

* fix(e2e): reference client.proxy in mid-conversation native providers test

EndpointsClient exposes the shared ProxyClient as .proxy and has never had a
.gateway attribute, so these two calls raised AttributeError at runtime and
failed the tests/e2e basedpyright zero-error gate for any PR touching e2e
files. Introduced in 23b5b7d199.

* test(e2e): cover 12 non-core LLM coverage registry cells

Raises Non-Core LLMs registry coverage from 24/50 to 36/50 (overall 51.9%
to 54.8%). Four cells were already asserted by existing tests and only
gain their covers marker (openai embeddings, openai image generation,
openai TTS, cohere rerank); one is dual-marked onto the existing
spend-tracking embeddings test rather than duplicated.

New tests: bedrock and vertex embeddings, streaming TTS (asserts chunked
transfer encoding so a buffered body cannot pass), audio transcriptions
via the realtime suite's wav fixture, moderations flag/pass pair, and
files list/retrieve in the batches suite.

Harness: e2e_http.upload generalized to any form model with a
file_content_type override (batches path unchanged), new stream_binary
primitive + BinaryStream for binary chunked responses, transcribe and
moderations client methods, file retrieve/list client methods.

* fix(e2e): close streamed TTS response on error paths and surface the error body

With stream=True a non-2xx response returned with the body unread, keeping
the socket checked out until garbage collection; the sibling
_streaming_outcome already consumes resp.text on error. The response now
closes on every path and BinaryStream carries a bounded error_body so a
failed stream call is triageable.

* test(e2e): assert streamed TTS response carries no content-length
This commit is contained in:
ryan-crabbe-berri 2026-07-21 17:43:41 -07:00 • committed by GitHub
parent e7b9357bc2
commit e967bc8c4f
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
12 changed files with 496 additions and 20 deletions

View file

@ -26,16 +26,24 @@ from e2e_http import (
)
from models import LiteLLMParamsBody
UPLOAD_FILENAME = "batch_input.jsonl"
class FileObject(BaseModel):
id: str
object: str | None = None
purpose: str | None = None
filename: str | None = None
bytes: int | None = None
status: str | None = None
created_at: int | None = None
class FileList(BaseModel):
object: str | None = None
data: list[FileObject] = []
class BatchObject(BaseModel):
id: str
object: str | None = None
@ -106,12 +114,30 @@ class BatchClient:
_files_path(provider),
headers=self.proxy.transport.bearer(key),
form=form,
filename="batch_input.jsonl",
filename=UPLOAD_FILENAME,
content=content,
params=ModelQuery(model=model),
response_type=FileObject,
)
def retrieve_file(
self, file_id: str, *, key: str, provider: str | None = None
) -> Result[FileObject]:
return self.proxy.transport.get(
f"{_files_path(provider)}/{file_id}",
headers=self.proxy.transport.bearer(key),
params=NoBody(),
response_type=FileObject,
)
def list_files(self, *, key: str, provider: str | None = None) -> Result[FileList]:
return self.proxy.transport.get(
_files_path(provider),
headers=self.proxy.transport.bearer(key),
params=NoBody(),
response_type=FileList,
)
def create_batch(
self, *, body: BatchCreateBody, key: str, provider: str | None = None
) -> StreamingResponse:

View file

@ -26,6 +26,7 @@ import pytest
from e2e_config import require_env, unique_marker
from batch_client import (
UPLOAD_FILENAME,
BatchClient,
BatchCreateBody,
BatchObject,
@ -511,6 +512,72 @@ class TestBatchFileContent:
)
class TestOpenAIFiles:
"""GET /v1/files (list) and GET /v1/files/{id} (retrieve) over the OpenAI route.
The proxy lists the OpenAI org's raw file ids, so the list case uploads a raw
(provider-routed) file whose id matches what list returns; retrieve re-encodes
the id it was called with, so the model-encoded upload round-trips unchanged.
"""
@pytest.mark.covers(
"llm.files.openai.list.nonstream.works",
exercised_on=["files"],
)
def test_uploaded_file_appears_in_list(
self, client: BatchClient, resources: ResourceManager, batch_deployments: None
) -> None:
key = resources.key()
file = unwrap(
client.upload_file(
content=render_jsonl(OPENAI_BATCH_MODEL),
form=FileUploadForm(purpose="batch"),
key=key,
provider="openai",
)
)
resources.defer(
quietly(lambda: client.delete_file(file.id, key=key, provider="openai"))
)
listed = unwrap(client.list_files(key=key))
assert listed.object is None or listed.object == "list", (
f"list envelope object={listed.object!r}"
)
match = next((entry for entry in listed.data if entry.id == file.id), None)
assert match is not None, f"uploaded file {file.id!r} absent from GET /v1/files"
assert match.purpose == "batch", (
f"listed file must round-trip the upload purpose, got {match.purpose!r}"
)
@pytest.mark.covers(
"llm.files.openai.retrieve.nonstream.works",
exercised_on=["files"],
)
def test_retrieve_round_trips_metadata(
self, client: BatchClient, resources: ResourceManager, batch_deployments: None
) -> None:
key = resources.key()
file = unwrap(
client.upload_file(
content=render_jsonl(OPENAI_BATCH_MODEL),
form=FileUploadForm(purpose="batch"),
model=OPENAI_BATCH_MODEL,
key=key,
)
)
resources.defer(quietly(lambda: client.delete_file(file.id, key=key)))
fetched = unwrap(client.retrieve_file(file.id, key=key))
assert fetched.id == file.id, "retrieve must echo the uploaded file id"
assert fetched.purpose == "batch", (
f"retrieve must round-trip purpose, got {fetched.purpose!r}"
)
assert fetched.filename == UPLOAD_FILENAME, (
f"retrieve must round-trip filename, got {fetched.filename!r}"
)
BATCH_RL_REQUEST_LINES = 3
BATCH_RL_RPM_LIMIT = 2

View file

@ -147,6 +147,32 @@ class StreamingResponse(BaseModel):
return "text/event-stream" in (self.content_type or "")
class BinaryStream(BaseModel):
"""Outcome of consuming a binary chunked response (e.g. TTS audio) as a stream.
Unlike StreamingResponse, which line-splits an SSE text body, this iterates the
raw bytes with iter_content and reports how many non-empty chunks arrived and
the total byte count, so a caller can assert customer-observable streaming
(multiple chunks, real bytes) without decoding the payload."""
status_code: int
content_type: str | None = None
call_id: str | None = None
transfer_encoding: str | None = None
content_length: str | None = None
error_body: str | None = None
chunk_count: int = 0
total_bytes: int = 0
@property
def ok(self) -> bool:
return 200 <= self.status_code < 300
@property
def chunked(self) -> bool:
return "chunked" in (self.transfer_encoding or "")
def _hdr(resp: requests.Response, name: str) -> str | None:
value = resp.headers.get(name)
return value if isinstance(value, str) else None
@ -430,16 +456,18 @@ def upload[R: BaseModel](
url: URL,
*,
headers: BaseModel,
form: FileUploadForm,
form: BaseModel,
filename: str,
content: bytes,
file_content_type: str = "application/jsonl",
params: BaseModel | None = None,
response_type: type[R],
timeout: float = 60.0,
) -> Result[R]:
"""Multipart POST for file uploads (/v1/files). Form fields come from `form`,
the file bytes are sent as the `file` part, and `params` carries any query
routing (e.g. ?model=). requests sets the multipart Content-Type itself."""
"""Multipart POST for file-bearing routes (/v1/files, /v1/audio/transcriptions).
Form fields come from `form`, the file bytes are sent as the `file` part with
`file_content_type`, and `params` carries any query routing (e.g. ?model=).
requests sets the multipart Content-Type itself."""
dumped: dict[str, object] = form.model_dump(by_alias=True, exclude_none=True)
data = {key: str(value) for key, value in dumped.items()}
try:
@ -448,7 +476,7 @@ def upload[R: BaseModel](
headers=_headers(headers),
params=_params(params),
data=data,
files={"file": (filename, content, "application/jsonl")},
files={"file": (filename, content, file_content_type)},
timeout=timeout,
)
except requests.RequestException as exc:
@ -456,6 +484,54 @@ def upload[R: BaseModel](
return _classify(resp, response_type)
def stream_binary(
url: URL,
*,
headers: BaseModel,
json: BaseModel,
chunk_size: int = 8192,
timeout: float = 60.0,
) -> BinaryStream:
"""POST that consumes a binary chunked response (e.g. TTS audio) as a stream,
counting non-empty chunks and total bytes with iter_content. A non-2xx status
short-circuits with the counts left at zero so the caller can fail loudly."""
try:
resp = requests.post(
str(url),
headers=_headers(headers),
json=json.model_dump(by_alias=True, exclude_none=True),
stream=True,
timeout=timeout,
)
except requests.RequestException as exc:
return BinaryStream(status_code=-1, error_body=str(exc)[:300])
with resp:
content_type = _hdr(resp, "content-type")
call_id = _hdr(resp, "x-litellm-call-id")
transfer_encoding = _hdr(resp, "transfer-encoding")
content_length = _hdr(resp, "content-length")
if not (200 <= resp.status_code < 300):
return BinaryStream(
status_code=resp.status_code,
content_type=content_type,
call_id=call_id,
transfer_encoding=transfer_encoding,
content_length=content_length,
error_body=resp.text[:300],
)
raw_chunks = cast("Iterator[bytes]", resp.iter_content(chunk_size=chunk_size))
chunks = tuple(chunk for chunk in raw_chunks if chunk)
return BinaryStream(
status_code=resp.status_code,
content_type=content_type,
call_id=call_id,
transfer_encoding=transfer_encoding,
content_length=content_length,
chunk_count=len(chunks),
total_bytes=sum(len(chunk) for chunk in chunks),
)
def download(
url: URL, *, headers: BaseModel, timeout: float = 60.0
) -> StreamingResponse:

View file

@ -15,7 +15,7 @@ from typing import Literal
from pydantic import BaseModel
from proxy_client import ProxyClient
from e2e_http import StreamingResponse
from e2e_http import BinaryStream, Result, StreamingResponse
from models import CacheControl, ChatMessage, LiteLLMParamsBody, RichMessage, TextBlock
__all__ = [
@ -110,6 +110,16 @@ class ImageRequest(BaseModel):
size: str = "1024x1024"
class TranscriptionForm(BaseModel):
model: str
response_format: str = "json"
class ModerationRequest(BaseModel):
model: str
input: str
class ResponsesOutputContent(BaseModel):
type: str | None = None
text: str | None = None
@ -213,6 +223,27 @@ class ImagesResult(BaseModel):
data: list[ImageItem] = []
class TranscriptionResult(BaseModel):
text: str = ""
class ModerationResultItem(BaseModel):
flagged: bool
categories: dict[str, bool] = {}
@property
def flagged_categories(self) -> tuple[str, ...]:
return tuple(name for name, hit in self.categories.items() if hit)
class ModerationResult(BaseModel):
results: list[ModerationResultItem] = []
@property
def first(self) -> ModerationResultItem | None:
return self.results[0] if self.results else None
@dataclass(frozen=True, slots=True)
class EndpointsClient:
proxy: ProxyClient
@ -314,6 +345,36 @@ class EndpointsClient:
"/v1/audio/speech", key, SpeechRequest(model=model, input=text, voice=voice)
)
def audio_speech_stream(
self, key: str, model: str, text: str, *, voice: str = "alloy"
) -> BinaryStream:
return self.proxy.transport.stream_binary(
"/v1/audio/speech",
headers=self.proxy.transport.bearer(key),
json=SpeechRequest(model=model, input=text, voice=voice),
)
def transcribe(
self, key: str, model: str, *, filename: str, content: bytes
) -> Result[TranscriptionResult]:
return self.proxy.transport.upload(
"/v1/audio/transcriptions",
headers=self.proxy.transport.bearer(key),
form=TranscriptionForm(model=model),
filename=filename,
content=content,
file_content_type="audio/wav",
response_type=TranscriptionResult,
)
def moderations(self, key: str, model: str, text: str) -> Result[ModerationResult]:
return self.proxy.transport.post(
"/v1/moderations",
headers=self.proxy.transport.bearer(key),
json=ModerationRequest(model=model, input=text),
response_type=ModerationResult,
)
def images(self, key: str, model: str, prompt: str) -> StreamingResponse:
return self._send(
"/v1/images/generations", key, ImageRequest(model=model, prompt=prompt)

View file

@ -1,8 +1,9 @@
"""Live e2e: POST /v1/audio/speech returns audio.
"""Live e2e: POST /v1/audio/speech returns audio, non-streamed and streamed.
Registers an OpenAI text-to-speech deployment at runtime and asserts the response
is an audio body (binary, not JSON). Migrated from
litellm-regression-tests/tests/test_inference_endpoints.py.
The non-streamed call asserts an audio (not JSON) body. The streamed call consumes
the response the way a player would and asserts customer-observable streaming:
chunked transfer encoding (a buffered body would carry a content-length) with
non-zero audio bytes.
"""
from __future__ import annotations
@ -19,6 +20,7 @@ pytestmark = pytest.mark.e2e
class TestAudioSpeech:
@pytest.mark.covers("llm.audio_speech.openai.basic.nonstream.works")
def test_audio_speech_returns_audio(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:
@ -38,3 +40,39 @@ class TestAudioSpeech:
f"/audio/speech content-type is not audio: {result.content_type!r}"
)
assert result.body, "/audio/speech returned an empty body"
@pytest.mark.covers("llm.audio_speech.openai.basic.stream.works")
def test_audio_speech_streams_audio_chunks(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:
model = f"e2e-speech-stream-{unique_marker()}"
model_id = endpoints_client.create_model(
model,
LiteLLMParamsBody(
model="openai/gpt-4o-mini-tts", api_key="os.environ/OPENAI_API_KEY"
),
)
resources.defer(lambda: endpoints_client.delete_model(model_id))
key = resources.key()
result = endpoints_client.audio_speech_stream(
key,
model,
"Streaming speech should arrive in several audio chunks so a client can "
"begin playback well before the whole clip has finished generating.",
)
assert result.ok, (
f"/audio/speech stream failed (status {result.status_code}); body={result.error_body}"
)
assert "audio" in (result.content_type or ""), (
f"/audio/speech content-type is not audio: {result.content_type!r}"
)
assert result.chunked, (
f"/audio/speech did not stream: transfer-encoding={result.transfer_encoding!r}, "
f"content-length={result.content_length!r} (a buffered body is not a stream)"
)
assert result.content_length is None, (
f"/audio/speech advertised content-length={result.content_length!r} on a "
f"streamed response (a buffered body is not a stream)"
)
assert result.total_bytes > 0, "/audio/speech stream returned no audio bytes"

View file

@ -0,0 +1,51 @@
"""Live e2e: POST /v1/audio/transcriptions turns speech into text.
Registers an OpenAI speech-to-text deployment at runtime and uploads a spoken
weather question (the realtime suite's 24kHz WAV fixture) as multipart, asserting
the returned transcript is non-empty and mentions the word it was asked about.
"""
from __future__ import annotations
from pathlib import Path
import pytest
from e2e_config import unique_marker
from e2e_http import unwrap
from endpoints_client import EndpointsClient
from lifecycle import ResourceManager
from models import LiteLLMParamsBody
pytestmark = pytest.mark.e2e
WEATHER_WAV = (
Path(__file__).resolve().parent / "realtime" / "fixtures" / "weather_question_24k.wav"
)
class TestAudioTranscriptions:
@pytest.mark.covers("llm.audio_transcriptions.openai.basic.nonstream.works")
def test_audio_transcriptions_returns_text(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:
model = f"e2e-transcribe-{unique_marker()}"
model_id = endpoints_client.create_model(
model,
LiteLLMParamsBody(
model="openai/gpt-4o-mini-transcribe", api_key="os.environ/OPENAI_API_KEY"
),
)
resources.defer(lambda: endpoints_client.delete_model(model_id))
key = resources.key()
result = unwrap(
endpoints_client.transcribe(
key, model, filename=WEATHER_WAV.name, content=WEATHER_WAV.read_bytes()
)
)
text = result.text.strip()
assert text, "/audio/transcriptions returned an empty transcript"
assert "weather" in text.lower(), (
f"transcript of a spoken weather question does not mention weather: {text!r}"
)

View file

@ -1,9 +1,9 @@
"""Live e2e: POST /embeddings returns a real vector.
"""Live e2e: POST /embeddings returns a real vector across OpenAI, Bedrock, Vertex.
Registers an OpenAI embedding deployment at runtime and asserts a non-empty,
non-zero vector came back. Migrated from
litellm-regression-tests/tests/test_inference_endpoints.py; the LIT-3167 guard in
tests/e2e/embeddings/ covers the Gemini embedding path.
Each test registers the deployment it needs at runtime (deleted on teardown) and
asserts a non-empty, non-zero vector came back. The LIT-3167 guard in
tests/e2e/embeddings/ covers the Gemini embedding path; embeddings cost tracking is
covered by tests/e2e/quota_management/spend_tracking/.
"""
from __future__ import annotations
@ -20,6 +20,7 @@ pytestmark = pytest.mark.e2e
class TestEmbeddingsEndpoint:
@pytest.mark.covers("llm.embeddings.openai.basic.nonstream.works")
def test_embeddings_returns_vector(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:
@ -40,3 +41,49 @@ class TestEmbeddingsEndpoint:
assert any(component != 0.0 for component in parsed.first_vector), (
f"embedding vector is all zeros: {result.body[:300]}"
)
@pytest.mark.covers("llm.embeddings.bedrock.basic.nonstream.works")
def test_bedrock_embeddings_returns_vector(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:
model = f"e2e-embeddings-bedrock-{unique_marker()}"
model_id = endpoints_client.create_model(
model,
LiteLLMParamsBody(
model="bedrock/amazon.titan-embed-text-v2:0", aws_region_name="us-west-2"
),
)
resources.defer(lambda: endpoints_client.delete_model(model_id))
key = resources.key()
result = endpoints_client.embeddings(key, model, "Say this is a test!")
require_successful_call(result)
parsed = EmbeddingsResult.model_validate_json(result.body)
assert parsed.first_vector, f"/embeddings returned no vector: {result.body[:300]}"
assert any(component != 0.0 for component in parsed.first_vector), (
f"embedding vector is all zeros: {result.body[:300]}"
)
@pytest.mark.covers("llm.embeddings.vertex.basic.nonstream.works")
def test_vertex_embeddings_returns_vector(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:
model = f"e2e-embeddings-vertex-{unique_marker()}"
model_id = endpoints_client.create_model(
model,
LiteLLMParamsBody(
model="vertex_ai/gemini-embedding-2",
vertex_project="os.environ/VERTEXAI_PROJECT",
vertex_location="us-central1",
),
)
resources.defer(lambda: endpoints_client.delete_model(model_id))
key = resources.key()
result = endpoints_client.embeddings(key, model, "Say this is a test!")
require_successful_call(result)
parsed = EmbeddingsResult.model_validate_json(result.body)
assert parsed.first_vector, f"/embeddings returned no vector: {result.body[:300]}"
assert any(component != 0.0 for component in parsed.first_vector), (
f"embedding vector is all zeros: {result.body[:300]}"
)

View file

@ -19,6 +19,7 @@ pytestmark = pytest.mark.e2e
class TestImageGeneration:
@pytest.mark.covers("llm.images_generations.openai.basic.nonstream.works")
def test_image_generation_returns_image(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:

View file

@ -0,0 +1,65 @@
"""Live e2e: POST /v1/moderations classifies content against the provider policy.
Registers OpenAI's omni moderation model at runtime and asserts the product
promise on both sides of the decision: clearly violent text comes back flagged
with at least one policy category tripped, and benign text comes back not flagged.
"""
from __future__ import annotations
import pytest
from e2e_config import unique_marker
from e2e_http import unwrap
from endpoints_client import EndpointsClient
from lifecycle import ResourceManager
from models import LiteLLMParamsBody
pytestmark = pytest.mark.e2e
VIOLENT_TEXT = "I am going to find you and kill you, and I will hurt everyone you love."
BENIGN_TEXT = "I enjoyed the sunny afternoon and a relaxing walk in the park today."
def _register_moderation_model(
endpoints_client: EndpointsClient, resources: ResourceManager
) -> str:
model = f"e2e-moderation-{unique_marker()}"
model_id = endpoints_client.create_model(
model,
LiteLLMParamsBody(
model="openai/omni-moderation-latest", api_key="os.environ/OPENAI_API_KEY"
),
)
resources.defer(lambda: endpoints_client.delete_model(model_id))
return model
class TestModerations:
@pytest.mark.covers("llm.moderations.openai.basic.nonstream.works")
def test_moderations_flags_violent_content(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:
model = _register_moderation_model(endpoints_client, resources)
key = resources.key()
result = unwrap(endpoints_client.moderations(key, model, VIOLENT_TEXT))
item = result.first
assert item is not None, f"/moderations returned no results: {result}"
assert item.flagged, f"violent text was not flagged: {item}"
assert item.flagged_categories, (
f"flagged result reported no true category: {item}"
)
def test_moderations_passes_benign_content(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:
model = _register_moderation_model(endpoints_client, resources)
key = resources.key()
result = unwrap(endpoints_client.moderations(key, model, BENIGN_TEXT))
item = result.first
assert item is not None, f"/moderations returned no results: {result}"
assert not item.flagged, (
f"benign text was flagged as {item.flagged_categories}: {item}"
)

View file

@ -26,6 +26,7 @@ DOCUMENTS = [
class TestRerank:
@pytest.mark.covers("llm.rerank.cohere.basic.nonstream.works")
def test_rerank_scores_top_n(
self, endpoints_client: EndpointsClient, resources: ResourceManager
) -> None:

View file

@ -194,6 +194,7 @@ def test_streaming_messages_via_responses_bridge_tracks_spend(
@pytest.mark.covers("quota_management.spend_tracking.embeddings.logs_cost")
@pytest.mark.covers("llm.embeddings.openai.basic.nonstream.cost_logged")
def test_embedding_writes_nonzero_spend_row(
client: SpendClient, scoped_key: str
) -> None:

View file

@ -16,7 +16,7 @@ import e2e_http
from e2e_http import (
URL,
AuthHeaders,
FileUploadForm,
BinaryStream,
ProbeResult,
Result,
StreamingResponse,
@ -32,6 +32,15 @@ class Transport(Protocol):
self, path: str, *, headers: BaseModel, json: BaseModel
) -> StreamingResponse: ...
def stream_binary(
self,
path: str,
*,
headers: BaseModel,
json: BaseModel,
chunk_size: int = 8192,
) -> BinaryStream: ...
def send(
self,
path: str,
@ -76,9 +85,10 @@ class Transport(Protocol):
path: str,
*,
headers: BaseModel,
form: FileUploadForm,
form: BaseModel,
filename: str,
content: bytes,
file_content_type: str = "application/jsonl",
params: BaseModel | None = None,
response_type: type[R],
) -> Result[R]: ...
@ -181,6 +191,22 @@ class HttpTransport:
self._url(path), headers=headers, json=json, timeout=self.request_timeout
)
def stream_binary(
self,
path: str,
*,
headers: BaseModel,
json: BaseModel,
chunk_size: int = 8192,
) -> BinaryStream:
return e2e_http.stream_binary(
self._url(path),
headers=headers,
json=json,
chunk_size=chunk_size,
timeout=self.request_timeout,
)
def send(
self,
path: str,
@ -212,9 +238,10 @@ class HttpTransport:
path: str,
*,
headers: BaseModel,
form: FileUploadForm,
form: BaseModel,
filename: str,
content: bytes,
file_content_type: str = "application/jsonl",
params: BaseModel | None = None,
response_type: type[R],
) -> Result[R]:
@ -224,6 +251,7 @@ class HttpTransport:
form=form,
filename=filename,
content=content,
file_content_type=file_content_type,
params=params,
response_type=response_type,
timeout=self.request_timeout,
@ -346,6 +374,18 @@ class SplitTransport:
) -> StreamingResponse:
return self._route(path).stream(path, headers=headers, json=json)
def stream_binary(
self,
path: str,
*,
headers: BaseModel,
json: BaseModel,
chunk_size: int = 8192,
) -> BinaryStream:
return self._route(path).stream_binary(
path, headers=headers, json=json, chunk_size=chunk_size
)
def send(
self,
path: str,
@ -367,9 +407,10 @@ class SplitTransport:
path: str,
*,
headers: BaseModel,
form: FileUploadForm,
form: BaseModel,
filename: str,
content: bytes,
file_content_type: str = "application/jsonl",
params: BaseModel | None = None,
response_type: type[R],
) -> Result[R]:
@ -379,6 +420,7 @@ class SplitTransport:
form=form,
filename=filename,
content=content,
file_content_type=file_content_type,
params=params,
response_type=response_type,
)