Merge pull request #37618 from BerriAI/litellm_lit_5870_passthrough_e2e_pins

test(e2e): pin openai_passthrough routing, cost logging, and file list isolation
This commit is contained in:
Mateo Wang 2026-08-20 19:22:39 -07:00 • committed by GitHub
commit 65b4ac012f
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 315 additions and 6 deletions

View file

@ -40,8 +40,15 @@ class FileObject(BaseModel):
class FileList(BaseModel):
"""GET /v1/files page. The cursors are modelled because they are part of the
page's isolation contract: they must address rows in `data`, never rows the
caller was not allowed to see."""
object: str | None = None
data: list[FileObject] = []
first_id: str | None = None
last_id: str | None = None
has_more: bool | None = None
class BatchObject(BaseModel):

View file

@ -572,6 +572,41 @@ class TestOpenAIFiles:
f"listed file must round-trip the upload purpose, got {match.purpose!r}"
)
@pytest.mark.covers(
"llm.files.openai.list_isolation.nonstream.works",
exercised_on=["files"],
)
def test_list_page_cursors_address_only_the_callers_own_files(
self, client: BatchClient, resources: ResourceManager
) -> None:
"""Pins GitHub issue #36087: a list page's pagination cursors must address
rows in that page.
The proxy fronts one shared provider account, so the upstream page is the
whole organization's. The gateway narrows `data` to the files the caller
owns, and `first_id` / `last_id` have to be narrowed with it: left as the
upstream org's, they hand any caller raw provider file ids belonging to
other tenants, which is the handle the file routes accept.
"""
key = resources.key(user_id=f"e2e-file-list-{unique_marker()}")
listed = unwrap(client.list_files(key=key))
expected_first = listed.data[0].id if listed.data else None
expected_last = listed.data[-1].id if listed.data else None
assert listed.first_id == expected_first, (
f"first_id {listed.first_id!r} is not the first row this caller can see "
f"({expected_first!r}); the page leaked another caller's file id"
)
assert listed.last_id == expected_last, (
f"last_id {listed.last_id!r} is not the last row this caller can see "
f"({expected_last!r}); the page leaked another caller's file id"
)
assert listed.has_more is not True, (
"the page advertises another page, but the proxy never forwards a cursor "
"upstream, so following it re-serves this same page forever"
)
@pytest.mark.covers(
"llm.files.openai.retrieve.nonstream.works",
exercised_on=["files"],

View file

@ -63,6 +63,7 @@
- {id: llm.responses.openai.input_validation.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: openai, capability: input_validation, streaming: nonstream, assertions: [works], source: "vendor strategy §9.9 / LIT-4778", rationale: "Responses missing/empty input and missing model are rejected"}
- {id: llm.responses.openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: stream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Streaming via /v1/responses"}
- {id: llm.responses.openai.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "response_api_endpoints/endpoints.py:26", rationale: "Cost logged on responses"}
- {id: llm.responses.openai.passthrough.stream.cost_logged, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: stream, assertions: [cost_logged], source: "test_passthrough_e2e.py", rationale: "A streamed POST /openai_passthrough/v1/responses is costed and keyed by the provider response id; it used to log a zero-cost row under a random id (GitHub issue #36523)"}
- {id: llm.responses.openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Tool calls via Responses API"}
- {id: llm.responses.openai.vision.nonstream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: vision, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Vision via Responses API"}
- {id: llm.responses.anthropic.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: anthropic, capability: basic, streaming: nonstream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Responses w/ Anthropic translation (smoke)"}

View file

@ -3,6 +3,7 @@
- {id: llm.embeddings.openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: embeddings, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_embeddings_endpoint_e2e.py:23", rationale: "Core endpoint, live vector response"}
- {id: llm.embeddings.openai.input_validation.nonstream.works, module: llm, tier: P1, subject_endpoint: embeddings, route: openai, capability: input_validation, streaming: nonstream, assertions: [works], source: "vendor strategy §9.3 / LIT-4778", rationale: "Missing model/input on /embeddings return client errors"}
- {id: llm.embeddings.openai.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: embeddings, route: openai, capability: basic, streaming: nonstream, assertions: [cost_logged], source: "SPEND_TRACKING_COVERAGE_MATRIX.md:34", rationale: "Cost tracking on embeddings"}
- {id: llm.embeddings.openai.passthrough.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: embeddings, route: openai, capability: basic, streaming: nonstream, assertions: [cost_logged], source: "test_passthrough_e2e.py", rationale: "POST /openai_passthrough/v1/embeddings is costed; the route wrote no spend row at all, so budgets never saw traffic OpenAI was billing for (GitHub issue #36646)"}
- {id: llm.embeddings.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: embeddings, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/azure/azure.py", rationale: "Azure embeddings via translation"}
- {id: llm.embeddings.bedrock.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: embeddings, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "llms/bedrock/embed/embedding.py", rationale: "Bedrock Titan embeddings"}
- {id: llm.embeddings.vertex.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: embeddings, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "vertex_embeddings/embedding_handler.py", rationale: "Vertex embeddings"}
@ -13,6 +14,7 @@
- {id: llm.batches.openai.cancel.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Batch cancel"}
- {id: llm.batches.openai.list.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Batch list envelope"}
- {id: llm.batches.openai.file_lifecycle.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "File upload/retrieve/delete for batch flow"}
- {id: llm.batches.openai.passthrough.nonstream.works, module: llm, tier: P1, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_passthrough_e2e.py", rationale: "GET /openai_passthrough/v1/batches relays OpenAI's own batch page; the dedicated prefix must not bind as a provider name on the /{provider}/v1/batches route (GitHub issue #36086)"}
- {id: llm.batches.openai_encoded.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py", rationale: "Encoded scenario lifecycle"}
- {id: llm.batches.openai_unified.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py", rationale: "Unified/managed-id scenario"}
- {id: llm.batches.openai_model_param.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py", rationale: "Model-param scenario"}
@ -29,6 +31,8 @@
- {id: llm.files.openai.retrieve.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "files_endpoints.py", rationale: "File retrieve by id"}
- {id: llm.files.openai.delete.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "files_endpoints.py", rationale: "File delete returns deleted=true"}
- {id: llm.files.openai.list.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "files_endpoints.py", rationale: "File list paginated"}
- {id: llm.files.openai.list_isolation.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "GET /v1/files pagination cursors address only rows the caller owns; on a shared provider account the upstream cursors otherwise hand out other tenants' raw provider file ids (GitHub issue #36087)"}
- {id: llm.files.openai.passthrough.nonstream.works, module: llm, tier: P1, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_passthrough_e2e.py", rationale: "POST/DELETE /openai_passthrough/v1/files relay OpenAI's own file object; the dedicated prefix must not bind as a provider name on the /{provider}/v1/files route (GitHub issue #36086)"}
- {id: llm.files.azure_openai.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:45", rationale: "Azure file upload managed backend"}
- {id: llm.files.vertex.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:52", rationale: "Vertex file upload to GCS"}
- {id: llm.files.bedrock.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:59", rationale: "Bedrock file upload to S3"}

View file

@ -52,7 +52,7 @@ class ResourceManager:
"""
client: ResourceClient
_cleanups: List[Callable[[], None]] = field(
_cleanups: List[Callable[[], object]] = field(
default_factory=list
) # mutable-ok: append-only teardown registry
@ -60,8 +60,11 @@ class ResourceManager:
"""No global setup needed today; present for lifecycle symmetry."""
return None
def defer(self, cleanup: Callable[[], None]) -> None:
"""Register a teardown action for any resource the test just created."""
def defer(self, cleanup: Callable[[], object]) -> None:
"""Register a teardown action for any resource the test just created.
Whatever the action returns is discarded, so a delete that answers with a
response model can be deferred directly."""
self._cleanups.append(cleanup)
def key(self, models: list[str] | None = None, user_id: str | None = "e2e-test-user") -> str:

View file

@ -15,7 +15,7 @@ from dataclasses import dataclass
from pydantic import BaseModel, Field
from proxy_client import ProxyClient
from e2e_http import Headers, StreamingResponse
from e2e_http import FileUploadForm, Headers, NoBody, Result, StreamingResponse
from models import ChatMessage
@ -113,6 +113,76 @@ class OpenAIChatBody(BaseModel):
max_completion_tokens: int = 64
class PassthroughFileObject(BaseModel):
id: str
object: str | None = None
purpose: str | None = None
filename: str | None = None
bytes: int | None = None
class PassthroughFileDeleted(BaseModel):
id: str
deleted: bool
class PassthroughListEntry(BaseModel):
id: str
class ResponsesUsage(BaseModel):
input_tokens: int
output_tokens: int
class ResponsesObject(BaseModel):
id: str
usage: ResponsesUsage | None = None
class ResponsesStreamEvent(BaseModel):
"""One SSE frame of a native Responses stream. Only the terminal frames carry a
`response`, so it stays optional and the deltas validate as themselves."""
type: str
response: ResponsesObject | None = None
def completed_responses_object(result: StreamingResponse) -> ResponsesObject | None:
"""The `response.completed` frame's response object, or None if the stream never
completed. Its `id` is what the spend row is keyed by on this route, and its
usage is what the row is priced from."""
events = (
ResponsesStreamEvent.model_validate_json(payload)
for payload in result.stream_events
)
completed = tuple(
event.response
for event in events
if event.type == "response.completed" and event.response is not None
)
return completed[-1] if completed else None
class OpenAIResponsesBody(BaseModel):
model: str
input: str
stream: bool = False
class OpenAIEmbeddingBody(BaseModel):
model: str
input: str
class PassthroughBatchList(BaseModel):
"""OpenAI's own batch page, relayed verbatim. `object` is required so a body
that is not an OpenAI list fails validation instead of passing vacuously."""
object: str
data: list[PassthroughListEntry]
def _tags_header(tags: list[str] | None) -> str | None:
return ",".join(tags) if tags else None
@ -196,6 +266,66 @@ class PassthroughClient:
stream=stream,
)
# ---- OpenAI file/batch routes under /openai_passthrough -------------
#
# Relayed to OpenAI untouched, which is the whole point of the prefix: the
# customer opts out of the gateway's managed-file handling here.
def openai_passthrough_upload_file(
self, key: str, *, content: bytes, filename: str
) -> Result[PassthroughFileObject]:
return self.proxy.transport.upload(
"/openai_passthrough/v1/files",
headers=self.proxy.transport.bearer(key),
form=FileUploadForm(purpose="batch"),
filename=filename,
content=content,
response_type=PassthroughFileObject,
)
def openai_passthrough_delete_file(
self, key: str, file_id: str
) -> Result[PassthroughFileDeleted]:
return self.proxy.transport.delete(
f"/openai_passthrough/v1/files/{file_id}",
headers=self.proxy.transport.bearer(key),
json=NoBody(),
response_type=PassthroughFileDeleted,
)
def openai_passthrough_list_batches(self, key: str) -> Result[PassthroughBatchList]:
return self.proxy.transport.get(
"/openai_passthrough/v1/batches",
headers=self.proxy.transport.bearer(key),
params=NoBody(),
response_type=PassthroughBatchList,
)
# ---- OpenAI inference routes under /openai_passthrough -------------
#
# Relayed to OpenAI verbatim, but still costed by the gateway: the customer
# budgets against this traffic, so a 200 that logs no spend is money the
# gateway never sees.
def openai_passthrough_responses(
self, key: str, model: str, text: str, *, stream: bool = False
) -> StreamingResponse:
return self.proxy.transport.send(
"/openai_passthrough/v1/responses",
headers=self.proxy.transport.bearer(key),
json=OpenAIResponsesBody(model=model, input=text, stream=stream),
stream=stream,
)
def openai_passthrough_embed(
self, key: str, model: str, text: str
) -> StreamingResponse:
return self.proxy.transport.send(
"/openai_passthrough/v1/embeddings",
headers=self.proxy.transport.bearer(key),
json=OpenAIEmbeddingBody(model=model, input=text),
)
def openai_chat(
self, key: str, model: str, text: str, *, max_completion_tokens: int = 64
) -> StreamingResponse:

View file

@ -13,8 +13,8 @@ A passthrough call returning non-2xx fails hard (never a skip); once it returns
import pytest
from e2e_config import unique_marker
from e2e_http import StreamingResponse, require_successful_call
from e2e_config import CHEAP_OPENAI_MODEL, unique_marker
from e2e_http import StreamingResponse, require_successful_call, unwrap
from lifecycle import ResourceManager
from models import KeyGenerateBody, SpendLogRow
from passthrough_client import (
@ -24,8 +24,11 @@ from passthrough_client import (
JsonSchema,
JsonSchemaProperty,
PassthroughClient,
completed_responses_object,
)
EMBEDDING_MODEL = "text-embedding-3-small"
pytestmark = pytest.mark.e2e
@ -210,3 +213,129 @@ class TestPassthroughModelAllowlist:
"a key restricted to gemini-2.5-flash must be denied a claude passthrough call, "
f"got {result.status_code}: {result.body[:300]}"
)
class TestOpenAIPassthroughPrefix:
"""The dedicated `/openai_passthrough` prefix must reach OpenAI, not be
swallowed by the provider-scoped `/{provider}/v1/...` routes.
The customer fronts OpenAI's own file and batch APIs through this prefix
precisely to opt out of the gateway's managed-file handling. `/v1/files` and
`/v1/batches` also answer `/{provider}/v1/files` and `/{provider}/v1/batches`,
so `openai_passthrough` used to bind as a provider name and the request died
inside the gateway with a provider-lookup error, never reaching OpenAI.
"""
@pytest.mark.covers("llm.files.openai.passthrough.nonstream.works")
def test_passthrough_prefix_uploads_a_file_to_openai(
self, client: PassthroughClient, resources: ResourceManager, scoped_key: str
) -> None:
"""Pins GitHub issue #36086: a file upload through the dedicated prefix
reaches OpenAI's file API instead of 500ing on a provider-name lookup."""
content = f'{{"marker":"{unique_marker()}"}}\n'.encode()
uploaded = unwrap(
client.openai_passthrough_upload_file(
scoped_key, content=content, filename="e2e-passthrough-batch.jsonl"
)
)
resources.defer(
lambda: client.openai_passthrough_delete_file(scoped_key, uploaded.id)
)
assert uploaded.object == "file", (
f"/openai_passthrough/v1/files did not relay OpenAI's file object: {uploaded}"
)
assert uploaded.purpose == "batch"
assert uploaded.bytes == len(content)
@pytest.mark.covers("llm.batches.openai.passthrough.nonstream.works")
def test_passthrough_prefix_lists_batches_from_openai(
self, client: PassthroughClient, scoped_key: str
) -> None:
"""Pins GitHub issue #36086 on the batches route: the dedicated prefix
relays OpenAI's own batch page instead of dying on the provider lookup."""
listed = unwrap(client.openai_passthrough_list_batches(scoped_key))
assert listed.object == "list", (
f"/openai_passthrough/v1/batches did not relay OpenAI's batch page: {listed}"
)
class TestOpenAIPassthroughSpend:
"""A call relayed to OpenAI's own endpoints must still be costed.
The customer routes native OpenAI traffic through `/openai_passthrough` and
budgets against it, so a call that returns 200 while logging no spend is money
the gateway never sees and a budget that never trips. Streamed Responses calls
and embeddings each used to land exactly that way, on separate code paths.
"""
@pytest.mark.covers("llm.responses.openai.passthrough.stream.cost_logged")
def test_streamed_responses_call_logs_its_cost(
self, client: PassthroughClient, scoped_key: str
) -> None:
"""Pins GitHub issue #36523: a streamed passthrough Responses call is billed
under the provider id the caller was served, never a $0 row under a random
id."""
result = client.openai_passthrough_responses(
scoped_key,
CHEAP_OPENAI_MODEL,
f"Say hi in one word. {unique_marker()}",
stream=True,
)
require_successful_call(result)
assert result.chunks > 0, "streamed responses passthrough produced no events"
completed = completed_responses_object(result)
assert completed is not None, (
f"the stream never delivered a response.completed frame, so there is no "
f"provider id to reconcile against: last events {result.stream_events[-3:]}"
)
assert completed.usage is not None, (
f"the completed response carried no usage to price from: {completed}"
)
rows = client.proxy.poll_logs_for_request_id(
completed.id, predicate=lambda rows: (rows[0].spend or 0) > 0
)
assert rows, (
f"no spend row for the response the customer was served ({completed.id}); "
"a streamed passthrough call OpenAI bills them for is invisible to the "
"gateway's own spend and budgets"
)
row = rows[0]
assert (row.spend or 0) > 0, f"streamed responses passthrough was not costed: {row}"
assert row.prompt_tokens == completed.usage.input_tokens, (
f"logged {row.prompt_tokens} prompt tokens, the response the customer read "
f"reported {completed.usage.input_tokens}"
)
assert row.completion_tokens == completed.usage.output_tokens, (
f"logged {row.completion_tokens} completion tokens, the response the customer "
f"read reported {completed.usage.output_tokens}"
)
@pytest.mark.covers("llm.embeddings.openai.passthrough.nonstream.cost_logged")
def test_embeddings_call_logs_its_cost(
self, client: PassthroughClient, scoped_key: str
) -> None:
"""Pins GitHub issue #36646: a passthrough embeddings call writes a priced
spend row instead of no row at all."""
result = client.openai_passthrough_embed(
scoped_key, EMBEDDING_MODEL, f"cost this sentence {unique_marker()}"
)
require_successful_call(result)
assert result.call_id, "embeddings passthrough returned no x-litellm-call-id"
rows = client.proxy.poll_logs_for_request_id(
result.call_id, predicate=lambda rows: (rows[0].spend or 0) > 0
)
assert rows, (
f"no spend row for embeddings call {result.call_id}; the customer is billed "
"by OpenAI for tokens the gateway never counted against their budget"
)
row = rows[0]
assert (row.spend or 0) > 0, f"embeddings passthrough was not costed: {row}"
assert (row.prompt_tokens or 0) > 0, (
f"the embeddings row logged no prompt tokens, so whatever cost it carries "
f"was not computed from the real usage: {row}"
)