mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
test(e2e): derive cache rates from first principles and ungate all_components cases
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
3e11c98676
commit
fc0cce553a
3 changed files with 22 additions and 19 deletions
|
|
@ -181,9 +181,6 @@
|
|||
"audio_output_tokens": 3
|
||||
},
|
||||
"requires_rates": [
|
||||
"cache_read_input_token_cost",
|
||||
"cache_creation_input_token_cost",
|
||||
"cache_creation_input_token_cost_above_1hr",
|
||||
"output_cost_per_reasoning_token",
|
||||
"input_cost_per_audio_token",
|
||||
"output_cost_per_audio_token"
|
||||
|
|
@ -193,7 +190,6 @@
|
|||
{
|
||||
"name": "all_components_fireworks",
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25},
|
||||
"requires_rates": ["cache_read_input_token_cost"],
|
||||
"wires": ["fireworks_chat"]
|
||||
},
|
||||
{
|
||||
|
|
@ -205,11 +201,6 @@
|
|||
"cache_write_1h_tokens": 10,
|
||||
"output_tokens": 25
|
||||
},
|
||||
"requires_rates": [
|
||||
"cache_read_input_token_cost",
|
||||
"cache_creation_input_token_cost",
|
||||
"cache_creation_input_token_cost_above_1hr"
|
||||
],
|
||||
"wires": ["anthropic_messages", "bedrock_converse"]
|
||||
},
|
||||
{
|
||||
|
|
@ -222,11 +213,6 @@
|
|||
"output_tokens": 25
|
||||
},
|
||||
"stream": true,
|
||||
"requires_rates": [
|
||||
"cache_read_input_token_cost",
|
||||
"cache_creation_input_token_cost",
|
||||
"cache_creation_input_token_cost_above_1hr"
|
||||
],
|
||||
"wires": ["anthropic_messages"]
|
||||
},
|
||||
{
|
||||
|
|
@ -240,7 +226,6 @@
|
|||
"audio_output_tokens": 3
|
||||
},
|
||||
"requires_rates": [
|
||||
"cache_read_input_token_cost",
|
||||
"output_cost_per_reasoning_token",
|
||||
"input_cost_per_audio_token",
|
||||
"output_cost_per_audio_token"
|
||||
|
|
@ -250,7 +235,7 @@
|
|||
{
|
||||
"name": "all_components_responses",
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25, "reasoning_tokens": 15},
|
||||
"requires_rates": ["cache_read_input_token_cost", "output_cost_per_reasoning_token"],
|
||||
"requires_rates": ["output_cost_per_reasoning_token"],
|
||||
"wires": ["openai_responses"]
|
||||
}
|
||||
]
|
||||
|
|
|
|||
|
|
@ -1686,6 +1686,13 @@
|
|||
"prompt_tokens": 100,
|
||||
"spend": 0.0216
|
||||
},
|
||||
"meta.llama4-maverick-17b-instruct-v1:0|all_components_anthropic": {
|
||||
"completion_tokens": 25,
|
||||
"input_cost": 0.0285,
|
||||
"output_cost": 0.0095,
|
||||
"prompt_tokens": 150,
|
||||
"spend": 0.038
|
||||
},
|
||||
"meta.llama4-maverick-17b-instruct-v1:0|basic": {
|
||||
"completion_tokens": 40,
|
||||
"input_cost": 0.0228,
|
||||
|
|
|
|||
|
|
@ -94,11 +94,22 @@ def expected_breakdown(model: FrontierModel, case: Case) -> ExpectedCost:
|
|||
or rates.output_cost_per_token
|
||||
or 0.0
|
||||
)
|
||||
write_rate: Final = (
|
||||
rates.cache_creation_input_token_cost
|
||||
if rates.cache_creation_input_token_cost is not None
|
||||
else in_rate
|
||||
)
|
||||
input_cost: Final = (
|
||||
u.fresh_input_tokens * in_rate
|
||||
+ u.cache_read_tokens * (rates.cache_read_input_token_cost or 0.0)
|
||||
+ u.cache_write_5m_tokens * (rates.cache_creation_input_token_cost or 0.0)
|
||||
+ u.cache_write_1h_tokens * (rates.cache_creation_input_token_cost_above_1hr or 0.0)
|
||||
+ u.cache_read_tokens
|
||||
* (rates.cache_read_input_token_cost if rates.cache_read_input_token_cost is not None else in_rate)
|
||||
+ u.cache_write_5m_tokens * write_rate
|
||||
+ u.cache_write_1h_tokens
|
||||
* (
|
||||
rates.cache_creation_input_token_cost_above_1hr
|
||||
if rates.cache_creation_input_token_cost_above_1hr is not None
|
||||
else write_rate
|
||||
)
|
||||
+ u.audio_input_tokens * (rates.input_cost_per_audio_token or 0.0)
|
||||
)
|
||||
output_cost: Final = (
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue