fix(budget_reservation): don't reserve backend tier rates for $0 deployments

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Devin AI 2026-08-27 05:07:28 +00:00
parent 3b50819468
commit afef2a0e2f
2 changed files with 127 additions and 0 deletions

View file

@ -57,6 +57,9 @@ _COUNTER_ENTITY_TYPES: Final[Mapping[str, str]] = {
}
_CUSTOM_RATE_KEYS: Final = ("input_cost_per_token", "output_cost_per_token")
class _CounterReservationUnavailable(Exception):
def __init__(
self,
@ -1182,10 +1185,26 @@ def _get_model_cost_infos(
return [base, *({**base, "tiered_pricing": table} for table in tiered_tables)]
def _deployment_declares_own_rates(deployment: Mapping[str, Any]) -> bool:
"""Whether a deployment's own config replaces the backend model's pricing.
A deployment that spells out per-token rates and no tier table of its own is
billed at those rates, so the backend model's published tier table must not be
used to estimate it: a deployment priced at 0 would otherwise reserve, and
reject on, budget it can never spend.
"""
sources: Final = (deployment.get("litellm_params") or {}, deployment.get("model_info") or {})
if any(source.get("tiered_pricing") for source in sources):
return False
return any(source.get(key) is not None for source in sources for key in _CUSTOM_RATE_KEYS)
def _deployment_tiered_pricing_table(
deployment: dict[str, Any],
llm_router: Router,
) -> list[dict] | None:
if _deployment_declares_own_rates(deployment):
return None
model_id: Final = deployment.get("model_info", {}).get("id")
backend_model: Final = deployment.get("litellm_params", {}).get("model")
if not isinstance(model_id, str) or not isinstance(backend_model, str):

View file

@ -30,6 +30,7 @@ from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessin
from litellm.proxy.spend_tracking.budget_reservation import (
TOKENIZE_OFF_EVENT_LOOP_MIN_CHARS,
_approximate_input_size,
estimate_request_input_cost,
estimate_request_max_cost,
get_budget_window_start,
invalidate_budget_reservation_counters,
@ -1115,6 +1116,113 @@ def test_reservation_uses_most_expensive_deployment_in_group():
assert estimated == pytest.approx(expected_expensive)
@pytest.mark.parametrize(
"deployment_overrides",
[
{"litellm_params": {"input_cost_per_token": 0, "output_cost_per_token": 0}},
{"model_info": {"input_cost_per_token": 0, "output_cost_per_token": 0}},
],
ids=["litellm_params", "model_info"],
)
def test_free_deployment_of_tiered_model_reserves_nothing(deployment_overrides):
"""A deployment priced at 0 on a model whose published entry carries a tier table
must not be estimated against that table. Spend tracking bills such a deployment at
its own rates, so reserving the published tier rate consumed, and rejected requests
against, budget the deployment can never spend."""
litellm_params = {
"model": "dashscope/qwen-plus-latest",
"api_key": "sk-fake",
**deployment_overrides.get("litellm_params", {}),
}
router = Router(
model_list=[
{
"model_name": "qwen-free",
"litellm_params": litellm_params,
**({"model_info": deployment_overrides["model_info"]} if "model_info" in deployment_overrides else {}),
}
]
)
request_body = {
"model": "qwen-free",
"messages": [{"role": "user", "content": "hello " * 100}],
"max_tokens": 500,
}
assert (
estimate_request_max_cost(
request_body=request_body,
route="/chat/completions",
llm_router=router,
)
== 0.0
)
assert (
estimate_request_input_cost(
request_body=request_body,
route="/chat/completions",
llm_router=router,
)
== 0.0
)
def test_priced_deployment_of_tiered_model_still_reserves_tier_rate():
"""The published tier table still governs a deployment that declares no rates of
its own, so the free-deployment carve-out cannot silently disable reservation."""
router = Router(
model_list=[
{
"model_name": "qwen-paid",
"litellm_params": {"model": "dashscope/qwen-plus-latest", "api_key": "sk-fake"},
}
]
)
estimated = estimate_request_max_cost(
request_body={
"model": "qwen-paid",
"messages": [{"role": "user", "content": "hello " * 100}],
"max_tokens": 500,
},
route="/chat/completions",
llm_router=router,
)
assert estimated is not None and estimated > 0
def test_deployment_declaring_own_tier_table_keeps_it():
"""A deployment that overrides pricing with its own tier table is estimated against
that table, not skipped as if it were unpriced."""
router = Router(
model_list=[
{
"model_name": "qwen-own-tiers",
"litellm_params": {
"model": "dashscope/qwen-plus-latest",
"api_key": "sk-fake",
"tiered_pricing": [
{"range": [0, 1000000], "input_cost_per_token": 1e-06, "output_cost_per_token": 2e-06}
],
},
}
]
)
estimated = estimate_request_max_cost(
request_body={
"model": "qwen-own-tiers",
"messages": [{"role": "user", "content": "hello " * 100}],
"max_tokens": 500,
},
route="/chat/completions",
llm_router=router,
)
assert estimated is not None and estimated > 0
@pytest.mark.asyncio
async def test_should_clamp_reservation_to_model_ceiling_when_caller_overrequests(
spend_counter_state,