mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
test(e2e): assert the real bill for the four fixed cost gaps
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
bb6d7bf243
commit
415b06f5ff
3 changed files with 21 additions and 21 deletions
|
|
@ -186,15 +186,12 @@ _WIRE_CAPS: Final[Mapping[str, frozenset[str]]] = MappingProxyType({
|
|||
}
|
||||
),
|
||||
"openai_responses": frozenset({"cache_read", "reasoning", "web_search", "response_model", "absent_usage"}),
|
||||
# Product gap: litellm hard-indexes message_delta["usage"] in
|
||||
# anthropic/chat/handler.py, so a usage-absent anthropic stream raises
|
||||
# KeyError; the real wire always carries it, so the case cannot be
|
||||
# represented.
|
||||
"anthropic_messages": frozenset({"cache_read", "cache_write_5m", "cache_write_1h", "web_search", "response_model"}),
|
||||
# Product gap: the gemini transform sets ModelResponse.model from the
|
||||
# request and drops the provider's modelVersion, so a response-model
|
||||
# override can never be priced on this wire.
|
||||
"gemini_generate": frozenset({"cache_read", "reasoning", "audio", "web_search", "absent_usage"}),
|
||||
"anthropic_messages": frozenset(
|
||||
{"cache_read", "cache_write_5m", "cache_write_1h", "web_search", "response_model", "absent_usage"}
|
||||
),
|
||||
"gemini_generate": frozenset(
|
||||
{"cache_read", "reasoning", "audio", "web_search", "response_model", "absent_usage"}
|
||||
),
|
||||
"together_chat": frozenset(
|
||||
{
|
||||
"cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio",
|
||||
|
|
@ -240,9 +237,6 @@ class Case:
|
|||
billed_web_search_calls: int = 0
|
||||
response_model_override: bool = False
|
||||
exact_spend: bool = True
|
||||
# stream_usage=absent on a wire with no proxy-side token recount means the
|
||||
# bill is exactly zero; asserted as such rather than skipped.
|
||||
expect_zero_bill: bool = False
|
||||
|
||||
def scenario(self, scenario_id: str, model: FrontierModel, text: str) -> Scenario:
|
||||
return Scenario(
|
||||
|
|
@ -359,10 +353,6 @@ def cases_for(model: FrontierModel) -> tuple[Case, ...]:
|
|||
stream=True,
|
||||
stream_usage="absent",
|
||||
exact_spend=False,
|
||||
# The responses surface bills only provider-reported usage;
|
||||
# with no usage in the stream the spend row is zero. Other
|
||||
# wires recount tokens proxy-side and bill a nonzero amount.
|
||||
expect_zero_bill=model.wire == "openai_responses",
|
||||
)
|
||||
if "absent_usage" in caps
|
||||
else None
|
||||
|
|
|
|||
|
|
@ -90,11 +90,6 @@ class TestTokenPricing:
|
|||
)
|
||||
assert row is not None, f"no spend row with a cost breakdown landed for {model.map_key}/{case.name}"
|
||||
|
||||
if not case.exact_spend and case.expect_zero_bill:
|
||||
# The provider reported no usage and this wire has no proxy-side
|
||||
# recount, so the bill is exactly zero.
|
||||
assert row.spend is not None and row.spend == 0, f"no-usage stream billed {row.spend}: {row}"
|
||||
return
|
||||
if not case.exact_spend:
|
||||
# stream_usage=absent: the provider reported no usage, so the row's
|
||||
# token counts are the proxy's own recount; only assert a bill landed.
|
||||
|
|
|
|||
|
|
@ -63,13 +63,18 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {
|
||||
"cache_creation_input_token_cost": 0.00033,
|
||||
"cache_creation_input_token_cost_above_1hr": 0.00044,
|
||||
"cache_read_input_token_cost": 1.4e-05,
|
||||
"input_cost_per_audio_token": 0.00066,
|
||||
"input_cost_per_token": 0.00014000000000000001,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 2000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_audio_token": 0.00077,
|
||||
"output_cost_per_reasoning_token": 0.00055,
|
||||
"output_cost_per_token": 0.00028000000000000003,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.03,
|
||||
|
|
@ -82,13 +87,18 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"fireworks_ai/kimi-k3": {
|
||||
"cache_creation_input_token_cost": 0.00033,
|
||||
"cache_creation_input_token_cost_above_1hr": 0.00044,
|
||||
"cache_read_input_token_cost": 1.2e-05,
|
||||
"input_cost_per_audio_token": 0.00066,
|
||||
"input_cost_per_token": 0.00012000000000000002,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 2000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_audio_token": 0.00077,
|
||||
"output_cost_per_reasoning_token": 0.00055,
|
||||
"output_cost_per_token": 0.00024000000000000003,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.03,
|
||||
|
|
@ -101,13 +111,18 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"fireworks_ai/qwen3p8-max": {
|
||||
"cache_creation_input_token_cost": 0.00033,
|
||||
"cache_creation_input_token_cost_above_1hr": 0.00044,
|
||||
"cache_read_input_token_cost": 1.3e-05,
|
||||
"input_cost_per_audio_token": 0.00066,
|
||||
"input_cost_per_token": 0.00013000000000000002,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 2000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_audio_token": 0.00077,
|
||||
"output_cost_per_reasoning_token": 0.00055,
|
||||
"output_cost_per_token": 0.00026000000000000003,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.03,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue