test(e2e): align completion, SAIL, and spend-log fixtures with supported contracts (#43902)

This commit is contained in:
yuneng-jiang 2026-09-30 14:21:22 -07:00 • committed by GitHub
parent 6b9766fa0c
commit e61733b170
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
6 changed files with 96 additions and 20 deletions

View file

@ -103,6 +103,7 @@
- {id: llm.messages.together_ai.multi_turn.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: together_ai, capability: multi_turn, streaming: nonstream, assertions: [works], source: "llm_translation/test_together_ai_e2e.py", rationale: "Together tool result round trip over /v1/messages"}
- {id: llm.chat_completions.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "service_tier flex, balanced and auto map to Sail completion windows and bill the matching price columns"}
- {id: llm.chat_completions.sail.service_tier.nonstream.rejects_unknown_tier, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [rejects_unknown_tier], source: "llm_translation/test_sail_e2e.py", rationale: "A service_tier Sail has no completion window for is a 400 without drop_params"}
- {id: llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [drops_unknown_tier_and_bills_asap], source: "llm_translation/test_sail_e2e.py", rationale: "An unknown service_tier under drop_params is dropped and billed at asap in both the cost header and spend log"}
- {id: llm.responses.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: responses, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "A caller metadata.completion_window of flex on /v1/responses bills Sail flex rates"}
- {id: llm.messages.sail.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: sail, capability: basic, streaming: nonstream, assertions: [works], source: "llm_translation/test_sail_e2e.py", rationale: "Sail over /v1/messages"}
- {id: llm.chat_completions.anthropic.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "llm_translation/test_conversational_matrix_e2e.py", rationale: "Anthropic over /chat/completions: cost header and spend row agree"}

View file

@ -2,7 +2,7 @@
The legacy text-completion endpoint (prompt-style, non-chat) is the second-busiest
route in production yet was previously uncovered; the rest of the "completions"
surface is chat only. Registers an OpenAI instruct deployment at runtime (deleted
surface is chat only. Registers an OpenAI chat deployment at runtime (deleted
on teardown), drives /v1/completions through the gateway with the real OpenAI SDK
(LIT-4577), and asserts real generated text came back so a regression that empties
the completion fails here.
@ -29,7 +29,7 @@ class TestCompletionsEndpoint:
model_id = proxy.create_model(
model,
LiteLLMParamsBody(
model="text-completion-openai/gpt-3.5-turbo-instruct",
model="openai/gpt-5.4-nano",
api_key="os.environ/OPENAI_API_KEY",
),
)
@ -40,7 +40,7 @@ class TestCompletionsEndpoint:
model=model,
prompt="Finish this sentence in a few words: the capital of France is",
max_tokens=32,
extra_body=NO_PROXY_CACHE,
extra_body={**NO_PROXY_CACHE, "reasoning_effort": "none"},
)
assert completion.choices, f"/v1/completions returned no choices: {completion!r}"
text = (completion.choices[0].text or "").strip()

View file

@ -13,7 +13,6 @@ from collections.abc import Mapping
from dataclasses import dataclass
from typing import Final, Literal
import openai
import pytest
from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS, unique_marker
from lifecycle import ResourceManager
@ -150,20 +149,29 @@ class TestSailChatCompletions:
)
_assert_spend_row_matches(proxy, key, header_cost)
@pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.rejects_unknown_tier")
def test_unknown_service_tier_is_rejected(
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
@pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap")
@pytest.mark.parametrize("service_tier", ["bogus", 5])
def test_unknown_service_tier_is_dropped_and_billed_asap(
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, service_tier: str | int
) -> None:
model, key = _register(proxy, resources)
with pytest.raises(openai.BadRequestError) as raised:
_ = _openai(sdk, key).chat.completions.create(
model=model,
messages=[{"role": "user", "content": PROMPT}],
max_completion_tokens=MAX_TOKENS,
extra_body={**NO_PROXY_CACHE, "service_tier": "bogus"},
)
assert "service_tier" in raised.value.message, f"400 does not name service_tier: {raised.value.message}"
raw: Final = _openai(sdk, key).chat.completions.with_raw_response.create(
model=model,
messages=[{"role": "user", "content": f"{PROMPT} {unique_marker()}"}],
max_completion_tokens=MAX_TOKENS,
extra_body={**NO_PROXY_CACHE, "service_tier": service_tier, "drop_params": True},
)
usage: Final = raw.parse().usage
assert usage is not None, "chat response carries no usage"
details: Final = usage.prompt_tokens_details
tokens: Final = _Tokens(
prompt=usage.prompt_tokens,
cached=(details.cached_tokens or 0) if details else 0,
completion=usage.completion_tokens,
)
header_cost: Final = _assert_billed_at("base", tokens, response_header(raw.headers, "x-litellm-response-cost"))
_assert_spend_row_matches(proxy, key, header_cost)
class TestSailResponses:

View file

@ -949,9 +949,14 @@ class GuardrailEntityMatch(BaseModel):
end: int
class GuardrailModeRecord(BaseModel):
tags: dict[str, str | list[str]] | None = None
default: str | list[str] | None = None
class GuardrailRunRecord(BaseModel):
guardrail_name: str | None = None
guardrail_mode: str | None = None
guardrail_mode: str | list[str] | GuardrailModeRecord | None = None
guardrail_status: str | None = None
guardrail_provider: str | None = None
masked_entity_count: dict[str, int] | None = None

View file

@ -12,6 +12,7 @@ monkeypatches anything.
from __future__ import annotations
import json
from collections.abc import Callable, Iterator, Mapping, Sequence
from dataclasses import dataclass
from types import MappingProxyType
@ -31,6 +32,7 @@ from e2e_http import (
wire_body,
without_retries,
)
from models import SpendLogs, SpendLogsPage
from pydantic import BaseModel, TypeAdapter
@ -217,3 +219,55 @@ class TestClassifyEmptyBody:
def test_body_that_is_not_json_is_still_a_validation_failure(self) -> None:
result: Final = classify(FakeJsonResponse(status_code=200, content=b"<html/>"), NoBody)
assert isinstance(result, ValidationError)
class TestSpendLogDecoding:
@pytest.mark.parametrize("paginated", [False, True])
@pytest.mark.parametrize(
"mode",
[
None,
"post_call",
["post_call"],
["pre_call", "post_call"],
{"tags": {"audit": ["post_call"]}, "default": "pre_call"},
],
)
def test_supported_guardrail_modes_preserve_neighbor_attribution_and_masked_response(
self, mode: object, paginated: bool
) -> None:
rows: Final = [
{
"request_id": "guarded-call",
"api_key": "scoped-key-hash",
"metadata": {"guardrail_information": [{"guardrail_mode": mode, "guardrail_status": "success"}]},
"response": {"content": "<CREDIT_CARD>"},
},
{"request_id": "health-call", "api_key": "litellm-health-check", "request_tags": ["litellm-health-check"]},
]
payload: Final = (
{"data": rows, "total": 2, "page": 1, "page_size": 100, "total_pages": 1} if paginated else rows
)
response: Final = FakeJsonResponse(status_code=200, content=json.dumps(payload).encode())
result: Final = classify(response, SpendLogsPage) if paginated else classify(response, SpendLogs)
assert isinstance(result, Success), result
decoded: Final = result.data.data if isinstance(result.data, SpendLogsPage) else result.data.root
assert [(row.request_id, row.api_key) for row in decoded] == [
("guarded-call", "scoped-key-hash"),
("health-call", "litellm-health-check"),
]
assert decoded[1].request_tags == ["litellm-health-check"]
assert decoded[0].response == {"content": "<CREDIT_CARD>"}
metadata: Final = decoded[0].metadata
assert metadata is not None and metadata.guardrail_information is not None
record: Final = metadata.guardrail_information[0]
assert record.model_dump(exclude_unset=True) == {"guardrail_mode": mode, "guardrail_status": "success"}
@pytest.mark.parametrize("mode", [5, [5], {"tags": {"audit": 5}}])
def test_malformed_guardrail_mode_remains_a_validation_failure(self, mode: object) -> None:
payload: Final = [{"metadata": {"guardrail_information": [{"guardrail_mode": mode}]}}]
result: Final = classify(FakeJsonResponse(status_code=200, content=json.dumps(payload).encode()), SpendLogs)
assert isinstance(result, ValidationError)
assert "guardrail_mode" in result.message

View file

@ -98,7 +98,7 @@ def test_sail_sync_chat_sends_the_tier_window(
assert _window(body) == window
@pytest.mark.parametrize("service_tier", ["scale", "standard", "asap", 5, ["flex"]])
@pytest.mark.parametrize("service_tier", ["bogus", "scale", "standard", "asap", 5, ["flex"]])
@pytest.mark.asyncio
async def test_sail_chat_rejects_a_tier_with_no_window_before_sending(
sail_env: None, chat_route: respx.Route, service_tier: object
@ -110,16 +110,24 @@ async def test_sail_chat_rejects_a_tier_with_no_window_before_sending(
assert not chat_route.called
@pytest.mark.parametrize("service_tier", ["scale", 5])
@pytest.mark.parametrize("service_tier", ["bogus", "scale", 5])
@pytest.mark.parametrize(("global_drop", "request_drop"), [(False, True), (True, False)])
@pytest.mark.asyncio
async def test_sail_chat_drops_an_unknown_tier_under_drop_params_and_bills_asap(
sail_env: None, chat_route: respx.Route, spend_capture: SpendCapture, service_tier: object
sail_env: None,
chat_route: respx.Route,
spend_capture: SpendCapture,
monkeypatch: pytest.MonkeyPatch,
service_tier: object,
global_drop: bool,
request_drop: bool,
) -> None:
monkeypatch.setattr(litellm, "drop_params", global_drop)
await litellm.acompletion(
model=MODEL,
messages=MESSAGES,
service_tier=service_tier,
drop_params=True,
drop_params=request_drop,
litellm_call_id=spend_capture.call_id,
)