mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
test(e2e): bill Sail windows that synchronous calls can still use (#44058)
Sail now rejects completion_window "flex" on synchronous requests with a 400 saying flex is only for background responses or Batch work. The chat flex case and the responses flex case have failed on every scheduled litellm-e2e run in builds 337, 340 and 341. The chat cases keep balanced and auto, and the responses case sends a caller metadata.completion_window of balanced, so both still prove the window reaches Sail and the bill uses that window's distinct rates
This commit is contained in:
parent
5e5882244a
commit
0da00d4b2e
2 changed files with 8 additions and 6 deletions
|
|
@ -101,10 +101,10 @@
|
|||
- {id: llm.messages.together_ai.basic.stream.works, module: llm, tier: P1, subject_endpoint: messages, route: together_ai, capability: basic, streaming: stream, assertions: [works], source: "llm_translation/test_together_ai_e2e.py", rationale: "Together over /v1/messages streaming"}
|
||||
- {id: llm.messages.together_ai.tool_use.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: together_ai, capability: tool_use, streaming: nonstream, assertions: [works], source: "llm_translation/test_together_ai_e2e.py", rationale: "Together tool calls over /v1/messages"}
|
||||
- {id: llm.messages.together_ai.multi_turn.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: together_ai, capability: multi_turn, streaming: nonstream, assertions: [works], source: "llm_translation/test_together_ai_e2e.py", rationale: "Together tool result round trip over /v1/messages"}
|
||||
- {id: llm.chat_completions.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "service_tier flex, balanced and auto map to Sail completion windows and bill the matching price columns"}
|
||||
- {id: llm.chat_completions.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "service_tier balanced and auto map to Sail completion windows and bill the matching price columns; Sail serves flex only to background responses and Batch"}
|
||||
- {id: llm.chat_completions.sail.service_tier.nonstream.rejects_unknown_tier, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [rejects_unknown_tier], source: "llm_translation/test_sail_e2e.py", rationale: "A service_tier Sail has no completion window for is a 400 without drop_params"}
|
||||
- {id: llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap, module: llm, tier: P1, subject_endpoint: chat_completions, route: sail, capability: service_tier, streaming: nonstream, assertions: [drops_unknown_tier_and_bills_asap], source: "llm_translation/test_sail_e2e.py", rationale: "An unknown service_tier under drop_params is dropped and billed at asap in both the cost header and spend log"}
|
||||
- {id: llm.responses.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: responses, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "A caller metadata.completion_window of flex on /v1/responses bills Sail flex rates"}
|
||||
- {id: llm.responses.sail.service_tier.nonstream.cost_logged, module: llm, tier: P1, subject_endpoint: responses, route: sail, capability: service_tier, streaming: nonstream, assertions: [cost_logged], source: "llm_translation/test_sail_e2e.py", rationale: "A caller metadata.completion_window of balanced on /v1/responses bills Sail balanced rates"}
|
||||
- {id: llm.messages.sail.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: sail, capability: basic, streaming: nonstream, assertions: [works], source: "llm_translation/test_sail_e2e.py", rationale: "Sail over /v1/messages"}
|
||||
- {id: llm.chat_completions.anthropic.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "llm_translation/test_conversational_matrix_e2e.py", rationale: "Anthropic over /chat/completions: cost header and spend row agree"}
|
||||
- {id: llm.chat_completions.anthropic.multi_turn.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: multi_turn, streaming: nonstream, assertions: [works], source: "llm_translation/test_conversational_matrix_e2e.py", rationale: "Anthropic tool result round trip over /chat/completions"}
|
||||
|
|
|
|||
|
|
@ -117,7 +117,7 @@ def _assert_spend_row_matches(proxy: ProxyClient, key: str, header_cost: float)
|
|||
class TestSailChatCompletions:
|
||||
@pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.cost_logged")
|
||||
@pytest.mark.parametrize(
|
||||
("service_tier", "billed_tier"), [("flex", "flex"), ("balanced", "balanced"), ("auto", "base")]
|
||||
("service_tier", "billed_tier"), [("balanced", "balanced"), ("auto", "base")]
|
||||
)
|
||||
def test_service_tier_bills_the_matching_completion_window(
|
||||
self,
|
||||
|
|
@ -176,7 +176,7 @@ class TestSailChatCompletions:
|
|||
|
||||
class TestSailResponses:
|
||||
@pytest.mark.covers("llm.responses.sail.service_tier.nonstream.cost_logged")
|
||||
def test_flex_completion_window_bills_flex_rates(
|
||||
def test_caller_completion_window_bills_its_rates(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model, key = _register(proxy, resources)
|
||||
|
|
@ -185,7 +185,7 @@ class TestSailResponses:
|
|||
model=model,
|
||||
input=f"{PROMPT} {unique_marker()}",
|
||||
max_output_tokens=MAX_TOKENS,
|
||||
metadata={"completion_window": "flex"},
|
||||
metadata={"completion_window": "balanced"},
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
usage: Final = raw.parse().usage
|
||||
|
|
@ -196,7 +196,9 @@ class TestSailResponses:
|
|||
completion=usage.output_tokens,
|
||||
)
|
||||
|
||||
header_cost: Final = _assert_billed_at("flex", tokens, response_header(raw.headers, "x-litellm-response-cost"))
|
||||
header_cost: Final = _assert_billed_at(
|
||||
"balanced", tokens, response_header(raw.headers, "x-litellm-response-cost")
|
||||
)
|
||||
_assert_spend_row_matches(proxy, key, header_cost)
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue