mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(cost): fall back to the model output rate when a tier omits one
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
3f64cbe41b
commit
c3e38a0b52
4 changed files with 75 additions and 4 deletions
|
|
@ -226,6 +226,8 @@ def _get_tiered_reasoning_rate(model_info: ModelInfo, usage: Usage) -> float | N
|
|||
tier: Final = _select_priced_tier(model_info=model_info, usage=usage)
|
||||
if tier is None:
|
||||
return None
|
||||
if "output_cost_per_reasoning_token" not in tier and "output_cost_per_token" not in tier:
|
||||
return None
|
||||
return tier_rate(tier, "output_cost_per_reasoning_token", "output_cost_per_token")
|
||||
|
||||
|
||||
|
|
@ -236,15 +238,23 @@ def _get_tiered_base_costs(model_info: ModelInfo, usage: Usage) -> tuple[float,
|
|||
Tiered pricing is all-or-nothing: one tier is picked from the request's input tokens
|
||||
and every token of the request is billed at that tier's rate. Rates the tier does not
|
||||
declare fall back to the tier's input rate, so a request never mixes tiers.
|
||||
|
||||
An output rate is the exception: a tier table that spells out only input rates would
|
||||
otherwise serve every completion for free, so the model's own output rate stands in.
|
||||
"""
|
||||
tier: Final = _select_priced_tier(model_info=model_info, usage=usage)
|
||||
if tier is None:
|
||||
return None
|
||||
|
||||
cache_creation_cost: Final = tier_rate(tier, "cache_creation_input_token_cost", "input_cost_per_token")
|
||||
completion_cost: Final = (
|
||||
tier_rate(tier, "output_cost_per_token")
|
||||
if "output_cost_per_token" in tier
|
||||
else _get_cost_per_unit(model_info, "output_cost_per_token") or 0.0
|
||||
)
|
||||
return (
|
||||
tier_rate(tier, "input_cost_per_token"),
|
||||
tier_rate(tier, "output_cost_per_token"),
|
||||
completion_cost,
|
||||
cache_creation_cost,
|
||||
tier_rate(tier, "cache_creation_input_token_cost_above_1hr", "cache_creation_input_token_cost")
|
||||
or cache_creation_cost,
|
||||
|
|
|
|||
|
|
@ -88,12 +88,18 @@ def _calculate_completion_cost(
|
|||
model_info: ModelInfo,
|
||||
tier: dict | None,
|
||||
) -> float:
|
||||
# A tier declaring no output rate falls back to the model's own, since a table spelling out
|
||||
# only input rates would otherwise serve every completion for free
|
||||
output_cost: Final = (
|
||||
tier_rate(tier, "output_cost_per_token")
|
||||
if tier is not None and "output_cost_per_token" in tier
|
||||
else float(model_info.get("output_cost_per_token") or 0.0)
|
||||
)
|
||||
if tier is not None:
|
||||
return (breakdown.completion_tokens * tier_rate(tier, "output_cost_per_token")) + (
|
||||
breakdown.reasoning_tokens * tier_rate(tier, "output_cost_per_reasoning_token", "output_cost_per_token")
|
||||
return (breakdown.completion_tokens * output_cost) + (
|
||||
breakdown.reasoning_tokens * (tier_rate(tier, "output_cost_per_reasoning_token") or output_cost)
|
||||
)
|
||||
|
||||
output_cost: Final = float(model_info.get("output_cost_per_token") or 0.0)
|
||||
reasoning_cost: Final = _flat_rate(model_info, "output_cost_per_reasoning_token", "output_cost_per_token")
|
||||
|
||||
return (breakdown.completion_tokens * output_cost) + (breakdown.reasoning_tokens * reasoning_cost)
|
||||
|
|
|
|||
|
|
@ -699,6 +699,41 @@ def test_generic_cost_per_token_tiered_pricing_is_all_or_nothing():
|
|||
litellm.model_cost.pop(model, None)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_tier_without_an_output_rate_bills_the_model_rate():
|
||||
"""Regression: a tier table that spells out only input rates served every completion for
|
||||
free, since a tier's missing output rate has no tier-level fallback to stand in for it."""
|
||||
model = "litellm-test-tiered-input-only"
|
||||
custom_llm_provider = "openrouter"
|
||||
litellm.register_model(
|
||||
{
|
||||
model: {
|
||||
"litellm_provider": custom_llm_provider,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2e-06,
|
||||
"output_cost_per_reasoning_token": 5e-06,
|
||||
"tiered_pricing": [{"range": [0, 128000], "input_cost_per_token": 1e-03}],
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
try:
|
||||
usage = Usage(
|
||||
prompt_tokens=13,
|
||||
completion_tokens=182,
|
||||
total_tokens=195,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(reasoning_tokens=100),
|
||||
)
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
assert round(prompt_cost, 12) == round(13 * 1e-03, 12)
|
||||
assert round(completion_cost, 12) == round((82 * 2e-06) + (100 * 5e-06), 12)
|
||||
finally:
|
||||
litellm.model_cost.pop(model, None)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_tiered_pricing_bills_reasoning_at_tier_rate():
|
||||
"""Regression: a tier's output_cost_per_reasoning_token must price reasoning tokens
|
||||
on the generic path and in the logged breakdown, not the tier's plain output rate."""
|
||||
|
|
|
|||
|
|
@ -324,6 +324,26 @@ class TestDashscopeCostCalculator:
|
|||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
|
||||
def test_dashscope_tier_without_an_output_rate_bills_the_model_rate(self):
|
||||
"""
|
||||
Regression: a tier declaring only an input rate served every completion for free,
|
||||
since a missing tier output rate had no tier-level fallback to stand in for it.
|
||||
"""
|
||||
litellm.model_cost["dashscope/qwen-input-only-tier-test"] = {
|
||||
"litellm_provider": "dashscope",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.6e-06,
|
||||
"tiered_pricing": [{"range": [0, 1000], "input_cost_per_token": 4e-07}],
|
||||
}
|
||||
|
||||
usage = Usage(prompt_tokens=500, completion_tokens=200)
|
||||
prompt_cost, completion_cost = dashscope_cost_per_token(
|
||||
model="qwen-input-only-tier-test", usage=usage
|
||||
)
|
||||
|
||||
assert math.isclose(prompt_cost, 500 * 4e-07, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, 200 * 1.6e-06, rel_tol=1e-10)
|
||||
|
||||
def test_dashscope_tiered_pricing_zero_input_falls_back_to_flat_rates(self):
|
||||
"""
|
||||
No tier can be selected without input tokens, so an empty-prompt request must
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue