From c3e38a0b528239119256732a4858de0e8fdeeaa2 Mon Sep 17 00:00:00 2001 From: mateo Date: Sat, 15 Aug 2026 03:05:13 +0000 Subject: [PATCH] fix(cost): fall back to the model output rate when a tier omits one Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../litellm_core_utils/llm_cost_calc/utils.py | 12 ++++++- litellm/llms/dashscope/cost_calculator.py | 12 +++++-- .../llm_cost_calc/test_llm_cost_calc_utils.py | 35 +++++++++++++++++++ .../test_dashscope_cost_calculator.py | 20 +++++++++++ 4 files changed, 75 insertions(+), 4 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 4a33424e88f..9d6ad8b6e39 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -226,6 +226,8 @@ def _get_tiered_reasoning_rate(model_info: ModelInfo, usage: Usage) -> float | N tier: Final = _select_priced_tier(model_info=model_info, usage=usage) if tier is None: return None + if "output_cost_per_reasoning_token" not in tier and "output_cost_per_token" not in tier: + return None return tier_rate(tier, "output_cost_per_reasoning_token", "output_cost_per_token") @@ -236,15 +238,23 @@ def _get_tiered_base_costs(model_info: ModelInfo, usage: Usage) -> tuple[float, Tiered pricing is all-or-nothing: one tier is picked from the request's input tokens and every token of the request is billed at that tier's rate. Rates the tier does not declare fall back to the tier's input rate, so a request never mixes tiers. + + An output rate is the exception: a tier table that spells out only input rates would + otherwise serve every completion for free, so the model's own output rate stands in. """ tier: Final = _select_priced_tier(model_info=model_info, usage=usage) if tier is None: return None cache_creation_cost: Final = tier_rate(tier, "cache_creation_input_token_cost", "input_cost_per_token") + completion_cost: Final = ( + tier_rate(tier, "output_cost_per_token") + if "output_cost_per_token" in tier + else _get_cost_per_unit(model_info, "output_cost_per_token") or 0.0 + ) return ( tier_rate(tier, "input_cost_per_token"), - tier_rate(tier, "output_cost_per_token"), + completion_cost, cache_creation_cost, tier_rate(tier, "cache_creation_input_token_cost_above_1hr", "cache_creation_input_token_cost") or cache_creation_cost, diff --git a/litellm/llms/dashscope/cost_calculator.py b/litellm/llms/dashscope/cost_calculator.py index e22d3e06be1..0c42f77e8ac 100644 --- a/litellm/llms/dashscope/cost_calculator.py +++ b/litellm/llms/dashscope/cost_calculator.py @@ -88,12 +88,18 @@ def _calculate_completion_cost( model_info: ModelInfo, tier: dict | None, ) -> float: + # A tier declaring no output rate falls back to the model's own, since a table spelling out + # only input rates would otherwise serve every completion for free + output_cost: Final = ( + tier_rate(tier, "output_cost_per_token") + if tier is not None and "output_cost_per_token" in tier + else float(model_info.get("output_cost_per_token") or 0.0) + ) if tier is not None: - return (breakdown.completion_tokens * tier_rate(tier, "output_cost_per_token")) + ( - breakdown.reasoning_tokens * tier_rate(tier, "output_cost_per_reasoning_token", "output_cost_per_token") + return (breakdown.completion_tokens * output_cost) + ( + breakdown.reasoning_tokens * (tier_rate(tier, "output_cost_per_reasoning_token") or output_cost) ) - output_cost: Final = float(model_info.get("output_cost_per_token") or 0.0) reasoning_cost: Final = _flat_rate(model_info, "output_cost_per_reasoning_token", "output_cost_per_token") return (breakdown.completion_tokens * output_cost) + (breakdown.reasoning_tokens * reasoning_cost) diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index bb70d09681e..821e9d23e89 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -699,6 +699,41 @@ def test_generic_cost_per_token_tiered_pricing_is_all_or_nothing(): litellm.model_cost.pop(model, None) +def test_generic_cost_per_token_tier_without_an_output_rate_bills_the_model_rate(): + """Regression: a tier table that spells out only input rates served every completion for + free, since a tier's missing output rate has no tier-level fallback to stand in for it.""" + model = "litellm-test-tiered-input-only" + custom_llm_provider = "openrouter" + litellm.register_model( + { + model: { + "litellm_provider": custom_llm_provider, + "mode": "chat", + "output_cost_per_token": 2e-06, + "output_cost_per_reasoning_token": 5e-06, + "tiered_pricing": [{"range": [0, 128000], "input_cost_per_token": 1e-03}], + } + } + ) + + try: + usage = Usage( + prompt_tokens=13, + completion_tokens=182, + total_tokens=195, + completion_tokens_details=CompletionTokensDetailsWrapper(reasoning_tokens=100), + ) + prompt_cost, completion_cost = generic_cost_per_token( + model=model, + usage=usage, + custom_llm_provider=custom_llm_provider, + ) + assert round(prompt_cost, 12) == round(13 * 1e-03, 12) + assert round(completion_cost, 12) == round((82 * 2e-06) + (100 * 5e-06), 12) + finally: + litellm.model_cost.pop(model, None) + + def test_generic_cost_per_token_tiered_pricing_bills_reasoning_at_tier_rate(): """Regression: a tier's output_cost_per_reasoning_token must price reasoning tokens on the generic path and in the logged breakdown, not the tier's plain output rate.""" diff --git a/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py b/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py index 20549fbd0fb..42577ec44e3 100644 --- a/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py +++ b/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py @@ -324,6 +324,26 @@ class TestDashscopeCostCalculator: assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10) + def test_dashscope_tier_without_an_output_rate_bills_the_model_rate(self): + """ + Regression: a tier declaring only an input rate served every completion for free, + since a missing tier output rate had no tier-level fallback to stand in for it. + """ + litellm.model_cost["dashscope/qwen-input-only-tier-test"] = { + "litellm_provider": "dashscope", + "mode": "chat", + "output_cost_per_token": 1.6e-06, + "tiered_pricing": [{"range": [0, 1000], "input_cost_per_token": 4e-07}], + } + + usage = Usage(prompt_tokens=500, completion_tokens=200) + prompt_cost, completion_cost = dashscope_cost_per_token( + model="qwen-input-only-tier-test", usage=usage + ) + + assert math.isclose(prompt_cost, 500 * 4e-07, rel_tol=1e-10) + assert math.isclose(completion_cost, 200 * 1.6e-06, rel_tol=1e-10) + def test_dashscope_tiered_pricing_zero_input_falls_back_to_flat_rates(self): """ No tier can be selected without input tokens, so an empty-prompt request must