mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
test(e2e): align completion, SAIL, and spend-log fixtures with supported contracts (#43902)
This commit is contained in:
parent
6b9766fa0c
commit
e61733b170
6 changed files with 96 additions and 20 deletions
|
|
@ -103,6 +103,7 @@
|
|||
- {id: llm.messages.together_ai.multi_turn.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: together_ai, capability: multi_turn, streaming: nonstream, assertions: [works], source: "llm_translation/test_together_ai_e2e.py", rationale: "Together tool result round trip over /v1/messages"}
|
||||
- {id: llm.chat_completions.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "service_tier flex, balanced and auto map to Sail completion windows and bill the matching price columns"}
|
||||
- {id: llm.chat_completions.sail.service_tier.nonstream.rejects_unknown_tier, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [rejects_unknown_tier], source: "llm_translation/test_sail_e2e.py", rationale: "A service_tier Sail has no completion window for is a 400 without drop_params"}
|
||||
- {id: llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [drops_unknown_tier_and_bills_asap], source: "llm_translation/test_sail_e2e.py", rationale: "An unknown service_tier under drop_params is dropped and billed at asap in both the cost header and spend log"}
|
||||
- {id: llm.responses.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: responses, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "A caller metadata.completion_window of flex on /v1/responses bills Sail flex rates"}
|
||||
- {id: llm.messages.sail.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: sail, capability: basic, streaming: nonstream, assertions: [works], source: "llm_translation/test_sail_e2e.py", rationale: "Sail over /v1/messages"}
|
||||
- {id: llm.chat_completions.anthropic.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "llm_translation/test_conversational_matrix_e2e.py", rationale: "Anthropic over /chat/completions: cost header and spend row agree"}
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
The legacy text-completion endpoint (prompt-style, non-chat) is the second-busiest
|
||||
route in production yet was previously uncovered; the rest of the "completions"
|
||||
surface is chat only. Registers an OpenAI instruct deployment at runtime (deleted
|
||||
surface is chat only. Registers an OpenAI chat deployment at runtime (deleted
|
||||
on teardown), drives /v1/completions through the gateway with the real OpenAI SDK
|
||||
(LIT-4577), and asserts real generated text came back so a regression that empties
|
||||
the completion fails here.
|
||||
|
|
@ -29,7 +29,7 @@ class TestCompletionsEndpoint:
|
|||
model_id = proxy.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="text-completion-openai/gpt-3.5-turbo-instruct",
|
||||
model="openai/gpt-5.4-nano",
|
||||
api_key="os.environ/OPENAI_API_KEY",
|
||||
),
|
||||
)
|
||||
|
|
@ -40,7 +40,7 @@ class TestCompletionsEndpoint:
|
|||
model=model,
|
||||
prompt="Finish this sentence in a few words: the capital of France is",
|
||||
max_tokens=32,
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
extra_body={**NO_PROXY_CACHE, "reasoning_effort": "none"},
|
||||
)
|
||||
assert completion.choices, f"/v1/completions returned no choices: {completion!r}"
|
||||
text = (completion.choices[0].text or "").strip()
|
||||
|
|
|
|||
|
|
@ -13,7 +13,6 @@ from collections.abc import Mapping
|
|||
from dataclasses import dataclass
|
||||
from typing import Final, Literal
|
||||
|
||||
import openai
|
||||
import pytest
|
||||
from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS, unique_marker
|
||||
from lifecycle import ResourceManager
|
||||
|
|
@ -150,20 +149,29 @@ class TestSailChatCompletions:
|
|||
)
|
||||
_assert_spend_row_matches(proxy, key, header_cost)
|
||||
|
||||
@pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.rejects_unknown_tier")
|
||||
def test_unknown_service_tier_is_rejected(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
@pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap")
|
||||
@pytest.mark.parametrize("service_tier", ["bogus", 5])
|
||||
def test_unknown_service_tier_is_dropped_and_billed_asap(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, service_tier: str | int
|
||||
) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
|
||||
with pytest.raises(openai.BadRequestError) as raised:
|
||||
_ = _openai(sdk, key).chat.completions.create(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": PROMPT}],
|
||||
max_completion_tokens=MAX_TOKENS,
|
||||
extra_body={**NO_PROXY_CACHE, "service_tier": "bogus"},
|
||||
)
|
||||
assert "service_tier" in raised.value.message, f"400 does not name service_tier: {raised.value.message}"
|
||||
raw: Final = _openai(sdk, key).chat.completions.with_raw_response.create(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": f"{PROMPT} {unique_marker()}"}],
|
||||
max_completion_tokens=MAX_TOKENS,
|
||||
extra_body={**NO_PROXY_CACHE, "service_tier": service_tier, "drop_params": True},
|
||||
)
|
||||
usage: Final = raw.parse().usage
|
||||
assert usage is not None, "chat response carries no usage"
|
||||
details: Final = usage.prompt_tokens_details
|
||||
tokens: Final = _Tokens(
|
||||
prompt=usage.prompt_tokens,
|
||||
cached=(details.cached_tokens or 0) if details else 0,
|
||||
completion=usage.completion_tokens,
|
||||
)
|
||||
header_cost: Final = _assert_billed_at("base", tokens, response_header(raw.headers, "x-litellm-response-cost"))
|
||||
_assert_spend_row_matches(proxy, key, header_cost)
|
||||
|
||||
|
||||
class TestSailResponses:
|
||||
|
|
|
|||
|
|
@ -949,9 +949,14 @@ class GuardrailEntityMatch(BaseModel):
|
|||
end: int
|
||||
|
||||
|
||||
class GuardrailModeRecord(BaseModel):
|
||||
tags: dict[str, str | list[str]] | None = None
|
||||
default: str | list[str] | None = None
|
||||
|
||||
|
||||
class GuardrailRunRecord(BaseModel):
|
||||
guardrail_name: str | None = None
|
||||
guardrail_mode: str | None = None
|
||||
guardrail_mode: str | list[str] | GuardrailModeRecord | None = None
|
||||
guardrail_status: str | None = None
|
||||
guardrail_provider: str | None = None
|
||||
masked_entity_count: dict[str, int] | None = None
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ monkeypatches anything.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections.abc import Callable, Iterator, Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
from types import MappingProxyType
|
||||
|
|
@ -31,6 +32,7 @@ from e2e_http import (
|
|||
wire_body,
|
||||
without_retries,
|
||||
)
|
||||
from models import SpendLogs, SpendLogsPage
|
||||
from pydantic import BaseModel, TypeAdapter
|
||||
|
||||
|
||||
|
|
@ -217,3 +219,55 @@ class TestClassifyEmptyBody:
|
|||
def test_body_that_is_not_json_is_still_a_validation_failure(self) -> None:
|
||||
result: Final = classify(FakeJsonResponse(status_code=200, content=b"<html/>"), NoBody)
|
||||
assert isinstance(result, ValidationError)
|
||||
|
||||
|
||||
class TestSpendLogDecoding:
|
||||
@pytest.mark.parametrize("paginated", [False, True])
|
||||
@pytest.mark.parametrize(
|
||||
"mode",
|
||||
[
|
||||
None,
|
||||
"post_call",
|
||||
["post_call"],
|
||||
["pre_call", "post_call"],
|
||||
{"tags": {"audit": ["post_call"]}, "default": "pre_call"},
|
||||
],
|
||||
)
|
||||
def test_supported_guardrail_modes_preserve_neighbor_attribution_and_masked_response(
|
||||
self, mode: object, paginated: bool
|
||||
) -> None:
|
||||
rows: Final = [
|
||||
{
|
||||
"request_id": "guarded-call",
|
||||
"api_key": "scoped-key-hash",
|
||||
"metadata": {"guardrail_information": [{"guardrail_mode": mode, "guardrail_status": "success"}]},
|
||||
"response": {"content": "<CREDIT_CARD>"},
|
||||
},
|
||||
{"request_id": "health-call", "api_key": "litellm-health-check", "request_tags": ["litellm-health-check"]},
|
||||
]
|
||||
payload: Final = (
|
||||
{"data": rows, "total": 2, "page": 1, "page_size": 100, "total_pages": 1} if paginated else rows
|
||||
)
|
||||
response: Final = FakeJsonResponse(status_code=200, content=json.dumps(payload).encode())
|
||||
result: Final = classify(response, SpendLogsPage) if paginated else classify(response, SpendLogs)
|
||||
|
||||
assert isinstance(result, Success), result
|
||||
decoded: Final = result.data.data if isinstance(result.data, SpendLogsPage) else result.data.root
|
||||
assert [(row.request_id, row.api_key) for row in decoded] == [
|
||||
("guarded-call", "scoped-key-hash"),
|
||||
("health-call", "litellm-health-check"),
|
||||
]
|
||||
assert decoded[1].request_tags == ["litellm-health-check"]
|
||||
assert decoded[0].response == {"content": "<CREDIT_CARD>"}
|
||||
metadata: Final = decoded[0].metadata
|
||||
assert metadata is not None and metadata.guardrail_information is not None
|
||||
record: Final = metadata.guardrail_information[0]
|
||||
assert record.model_dump(exclude_unset=True) == {"guardrail_mode": mode, "guardrail_status": "success"}
|
||||
|
||||
@pytest.mark.parametrize("mode", [5, [5], {"tags": {"audit": 5}}])
|
||||
def test_malformed_guardrail_mode_remains_a_validation_failure(self, mode: object) -> None:
|
||||
payload: Final = [{"metadata": {"guardrail_information": [{"guardrail_mode": mode}]}}]
|
||||
result: Final = classify(FakeJsonResponse(status_code=200, content=json.dumps(payload).encode()), SpendLogs)
|
||||
|
||||
assert isinstance(result, ValidationError)
|
||||
assert "guardrail_mode" in result.message
|
||||
|
|
|
|||
|
|
@ -98,7 +98,7 @@ def test_sail_sync_chat_sends_the_tier_window(
|
|||
assert _window(body) == window
|
||||
|
||||
|
||||
@pytest.mark.parametrize("service_tier", ["scale", "standard", "asap", 5, ["flex"]])
|
||||
@pytest.mark.parametrize("service_tier", ["bogus", "scale", "standard", "asap", 5, ["flex"]])
|
||||
@pytest.mark.asyncio
|
||||
async def test_sail_chat_rejects_a_tier_with_no_window_before_sending(
|
||||
sail_env: None, chat_route: respx.Route, service_tier: object
|
||||
|
|
@ -110,16 +110,24 @@ async def test_sail_chat_rejects_a_tier_with_no_window_before_sending(
|
|||
assert not chat_route.called
|
||||
|
||||
|
||||
@pytest.mark.parametrize("service_tier", ["scale", 5])
|
||||
@pytest.mark.parametrize("service_tier", ["bogus", "scale", 5])
|
||||
@pytest.mark.parametrize(("global_drop", "request_drop"), [(False, True), (True, False)])
|
||||
@pytest.mark.asyncio
|
||||
async def test_sail_chat_drops_an_unknown_tier_under_drop_params_and_bills_asap(
|
||||
sail_env: None, chat_route: respx.Route, spend_capture: SpendCapture, service_tier: object
|
||||
sail_env: None,
|
||||
chat_route: respx.Route,
|
||||
spend_capture: SpendCapture,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
service_tier: object,
|
||||
global_drop: bool,
|
||||
request_drop: bool,
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "drop_params", global_drop)
|
||||
await litellm.acompletion(
|
||||
model=MODEL,
|
||||
messages=MESSAGES,
|
||||
service_tier=service_tier,
|
||||
drop_params=True,
|
||||
drop_params=request_drop,
|
||||
litellm_call_id=spend_capture.call_id,
|
||||
)
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue