From e61733b1701d231cad735e8720c3bfeab47147a3 Mon Sep 17 00:00:00 2001
From: yuneng-jiang
Date: Wed, 30 Sep 2026 14:21:22 -0700
Subject: [PATCH] test(e2e): align completion, SAIL, and spend-log fixtures
with supported contracts (#43902)
---
.../coverage_registry/llm_conversational.yaml | 1 +
.../test_completions_endpoint_e2e.py | 6 +--
tests/e2e/llm_translation/test_sail_e2e.py | 32 ++++++-----
tests/e2e/models.py | 7 ++-
tests/e2e/test_e2e_http.py | 54 +++++++++++++++++++
.../chat/test_sail_chat_transformation.py | 16 ++++--
6 files changed, 96 insertions(+), 20 deletions(-)
diff --git a/tests/e2e/coverage_registry/llm_conversational.yaml b/tests/e2e/coverage_registry/llm_conversational.yaml
index 8de8d5875b4..61f3be34a43 100644
--- a/tests/e2e/coverage_registry/llm_conversational.yaml
+++ b/tests/e2e/coverage_registry/llm_conversational.yaml
@@ -103,6 +103,7 @@
- {id: llm.messages.together_ai.multi_turn.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: together_ai, capability: multi_turn, streaming: nonstream, assertions: [works], source: "llm_translation/test_together_ai_e2e.py", rationale: "Together tool result round trip over /v1/messages"}
- {id: llm.chat_completions.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "service_tier flex, balanced and auto map to Sail completion windows and bill the matching price columns"}
- {id: llm.chat_completions.sail.service_tier.nonstream.rejects_unknown_tier, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [rejects_unknown_tier], source: "llm_translation/test_sail_e2e.py", rationale: "A service_tier Sail has no completion window for is a 400 without drop_params"}
+- {id: llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [drops_unknown_tier_and_bills_asap], source: "llm_translation/test_sail_e2e.py", rationale: "An unknown service_tier under drop_params is dropped and billed at asap in both the cost header and spend log"}
- {id: llm.responses.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: responses, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "A caller metadata.completion_window of flex on /v1/responses bills Sail flex rates"}
- {id: llm.messages.sail.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: sail, capability: basic, streaming: nonstream, assertions: [works], source: "llm_translation/test_sail_e2e.py", rationale: "Sail over /v1/messages"}
- {id: llm.chat_completions.anthropic.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "llm_translation/test_conversational_matrix_e2e.py", rationale: "Anthropic over /chat/completions: cost header and spend row agree"}
diff --git a/tests/e2e/llm_translation/test_completions_endpoint_e2e.py b/tests/e2e/llm_translation/test_completions_endpoint_e2e.py
index 63fcee3ce36..6202dada599 100644
--- a/tests/e2e/llm_translation/test_completions_endpoint_e2e.py
+++ b/tests/e2e/llm_translation/test_completions_endpoint_e2e.py
@@ -2,7 +2,7 @@
The legacy text-completion endpoint (prompt-style, non-chat) is the second-busiest
route in production yet was previously uncovered; the rest of the "completions"
-surface is chat only. Registers an OpenAI instruct deployment at runtime (deleted
+surface is chat only. Registers an OpenAI chat deployment at runtime (deleted
on teardown), drives /v1/completions through the gateway with the real OpenAI SDK
(LIT-4577), and asserts real generated text came back so a regression that empties
the completion fails here.
@@ -29,7 +29,7 @@ class TestCompletionsEndpoint:
model_id = proxy.create_model(
model,
LiteLLMParamsBody(
- model="text-completion-openai/gpt-3.5-turbo-instruct",
+ model="openai/gpt-5.4-nano",
api_key="os.environ/OPENAI_API_KEY",
),
)
@@ -40,7 +40,7 @@ class TestCompletionsEndpoint:
model=model,
prompt="Finish this sentence in a few words: the capital of France is",
max_tokens=32,
- extra_body=NO_PROXY_CACHE,
+ extra_body={**NO_PROXY_CACHE, "reasoning_effort": "none"},
)
assert completion.choices, f"/v1/completions returned no choices: {completion!r}"
text = (completion.choices[0].text or "").strip()
diff --git a/tests/e2e/llm_translation/test_sail_e2e.py b/tests/e2e/llm_translation/test_sail_e2e.py
index 9c714544d6e..cf662afea90 100644
--- a/tests/e2e/llm_translation/test_sail_e2e.py
+++ b/tests/e2e/llm_translation/test_sail_e2e.py
@@ -13,7 +13,6 @@ from collections.abc import Mapping
from dataclasses import dataclass
from typing import Final, Literal
-import openai
import pytest
from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS, unique_marker
from lifecycle import ResourceManager
@@ -150,20 +149,29 @@ class TestSailChatCompletions:
)
_assert_spend_row_matches(proxy, key, header_cost)
- @pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.rejects_unknown_tier")
- def test_unknown_service_tier_is_rejected(
- self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
+ @pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap")
+ @pytest.mark.parametrize("service_tier", ["bogus", 5])
+ def test_unknown_service_tier_is_dropped_and_billed_asap(
+ self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, service_tier: str | int
) -> None:
model, key = _register(proxy, resources)
- with pytest.raises(openai.BadRequestError) as raised:
- _ = _openai(sdk, key).chat.completions.create(
- model=model,
- messages=[{"role": "user", "content": PROMPT}],
- max_completion_tokens=MAX_TOKENS,
- extra_body={**NO_PROXY_CACHE, "service_tier": "bogus"},
- )
- assert "service_tier" in raised.value.message, f"400 does not name service_tier: {raised.value.message}"
+ raw: Final = _openai(sdk, key).chat.completions.with_raw_response.create(
+ model=model,
+ messages=[{"role": "user", "content": f"{PROMPT} {unique_marker()}"}],
+ max_completion_tokens=MAX_TOKENS,
+ extra_body={**NO_PROXY_CACHE, "service_tier": service_tier, "drop_params": True},
+ )
+ usage: Final = raw.parse().usage
+ assert usage is not None, "chat response carries no usage"
+ details: Final = usage.prompt_tokens_details
+ tokens: Final = _Tokens(
+ prompt=usage.prompt_tokens,
+ cached=(details.cached_tokens or 0) if details else 0,
+ completion=usage.completion_tokens,
+ )
+ header_cost: Final = _assert_billed_at("base", tokens, response_header(raw.headers, "x-litellm-response-cost"))
+ _assert_spend_row_matches(proxy, key, header_cost)
class TestSailResponses:
diff --git a/tests/e2e/models.py b/tests/e2e/models.py
index 4f8f6926095..8dccec8d9e1 100644
--- a/tests/e2e/models.py
+++ b/tests/e2e/models.py
@@ -949,9 +949,14 @@ class GuardrailEntityMatch(BaseModel):
end: int
+class GuardrailModeRecord(BaseModel):
+ tags: dict[str, str | list[str]] | None = None
+ default: str | list[str] | None = None
+
+
class GuardrailRunRecord(BaseModel):
guardrail_name: str | None = None
- guardrail_mode: str | None = None
+ guardrail_mode: str | list[str] | GuardrailModeRecord | None = None
guardrail_status: str | None = None
guardrail_provider: str | None = None
masked_entity_count: dict[str, int] | None = None
diff --git a/tests/e2e/test_e2e_http.py b/tests/e2e/test_e2e_http.py
index 7201da84924..e1c4145de8e 100644
--- a/tests/e2e/test_e2e_http.py
+++ b/tests/e2e/test_e2e_http.py
@@ -12,6 +12,7 @@ monkeypatches anything.
from __future__ import annotations
+import json
from collections.abc import Callable, Iterator, Mapping, Sequence
from dataclasses import dataclass
from types import MappingProxyType
@@ -31,6 +32,7 @@ from e2e_http import (
wire_body,
without_retries,
)
+from models import SpendLogs, SpendLogsPage
from pydantic import BaseModel, TypeAdapter
@@ -217,3 +219,55 @@ class TestClassifyEmptyBody:
def test_body_that_is_not_json_is_still_a_validation_failure(self) -> None:
result: Final = classify(FakeJsonResponse(status_code=200, content=b""), NoBody)
assert isinstance(result, ValidationError)
+
+
+class TestSpendLogDecoding:
+ @pytest.mark.parametrize("paginated", [False, True])
+ @pytest.mark.parametrize(
+ "mode",
+ [
+ None,
+ "post_call",
+ ["post_call"],
+ ["pre_call", "post_call"],
+ {"tags": {"audit": ["post_call"]}, "default": "pre_call"},
+ ],
+ )
+ def test_supported_guardrail_modes_preserve_neighbor_attribution_and_masked_response(
+ self, mode: object, paginated: bool
+ ) -> None:
+ rows: Final = [
+ {
+ "request_id": "guarded-call",
+ "api_key": "scoped-key-hash",
+ "metadata": {"guardrail_information": [{"guardrail_mode": mode, "guardrail_status": "success"}]},
+ "response": {"content": ""},
+ },
+ {"request_id": "health-call", "api_key": "litellm-health-check", "request_tags": ["litellm-health-check"]},
+ ]
+ payload: Final = (
+ {"data": rows, "total": 2, "page": 1, "page_size": 100, "total_pages": 1} if paginated else rows
+ )
+ response: Final = FakeJsonResponse(status_code=200, content=json.dumps(payload).encode())
+ result: Final = classify(response, SpendLogsPage) if paginated else classify(response, SpendLogs)
+
+ assert isinstance(result, Success), result
+ decoded: Final = result.data.data if isinstance(result.data, SpendLogsPage) else result.data.root
+ assert [(row.request_id, row.api_key) for row in decoded] == [
+ ("guarded-call", "scoped-key-hash"),
+ ("health-call", "litellm-health-check"),
+ ]
+ assert decoded[1].request_tags == ["litellm-health-check"]
+ assert decoded[0].response == {"content": ""}
+ metadata: Final = decoded[0].metadata
+ assert metadata is not None and metadata.guardrail_information is not None
+ record: Final = metadata.guardrail_information[0]
+ assert record.model_dump(exclude_unset=True) == {"guardrail_mode": mode, "guardrail_status": "success"}
+
+ @pytest.mark.parametrize("mode", [5, [5], {"tags": {"audit": 5}}])
+ def test_malformed_guardrail_mode_remains_a_validation_failure(self, mode: object) -> None:
+ payload: Final = [{"metadata": {"guardrail_information": [{"guardrail_mode": mode}]}}]
+ result: Final = classify(FakeJsonResponse(status_code=200, content=json.dumps(payload).encode()), SpendLogs)
+
+ assert isinstance(result, ValidationError)
+ assert "guardrail_mode" in result.message
diff --git a/tests/unit/llms/sail/chat/test_sail_chat_transformation.py b/tests/unit/llms/sail/chat/test_sail_chat_transformation.py
index a42fb1074a0..c7b1a77343a 100644
--- a/tests/unit/llms/sail/chat/test_sail_chat_transformation.py
+++ b/tests/unit/llms/sail/chat/test_sail_chat_transformation.py
@@ -98,7 +98,7 @@ def test_sail_sync_chat_sends_the_tier_window(
assert _window(body) == window
-@pytest.mark.parametrize("service_tier", ["scale", "standard", "asap", 5, ["flex"]])
+@pytest.mark.parametrize("service_tier", ["bogus", "scale", "standard", "asap", 5, ["flex"]])
@pytest.mark.asyncio
async def test_sail_chat_rejects_a_tier_with_no_window_before_sending(
sail_env: None, chat_route: respx.Route, service_tier: object
@@ -110,16 +110,24 @@ async def test_sail_chat_rejects_a_tier_with_no_window_before_sending(
assert not chat_route.called
-@pytest.mark.parametrize("service_tier", ["scale", 5])
+@pytest.mark.parametrize("service_tier", ["bogus", "scale", 5])
+@pytest.mark.parametrize(("global_drop", "request_drop"), [(False, True), (True, False)])
@pytest.mark.asyncio
async def test_sail_chat_drops_an_unknown_tier_under_drop_params_and_bills_asap(
- sail_env: None, chat_route: respx.Route, spend_capture: SpendCapture, service_tier: object
+ sail_env: None,
+ chat_route: respx.Route,
+ spend_capture: SpendCapture,
+ monkeypatch: pytest.MonkeyPatch,
+ service_tier: object,
+ global_drop: bool,
+ request_drop: bool,
) -> None:
+ monkeypatch.setattr(litellm, "drop_params", global_drop)
await litellm.acompletion(
model=MODEL,
messages=MESSAGES,
service_tier=service_tier,
- drop_params=True,
+ drop_params=request_drop,
litellm_call_id=spend_capture.call_id,
)