fix(proxy): bill tier-only deployments instead of $0

Route cost calculation to the deployment's router_model_id entry when it carries tiered_pricing but no flat per-token rate, so models like dashscope/qwen3.7-plus are billed via their tier table rather than the pricing-stripped shared alias.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Shivam Rawat 2026-07-11 13:47:56 -07:00
parent c52d23e171
commit 65e2432a37
2 changed files with 109 additions and 1 deletions

View file

@ -760,7 +760,11 @@ def _select_model_name_for_cost_calc(
if custom_pricing is True:
if router_model_id is not None and router_model_id in litellm.model_cost:
entry = litellm.model_cost[router_model_id]
if entry.get("input_cost_per_token") is not None or entry.get("input_cost_per_second") is not None:
if (
entry.get("input_cost_per_token") is not None
or entry.get("input_cost_per_second") is not None
or entry.get("tiered_pricing") is not None
):
return_model = router_model_id
else:
return_model = model

View file

@ -1000,6 +1000,110 @@ def test_per_request_custom_pricing_with_router():
assert "gpt-3.5-turbo" in selected
def test_tiered_pricing_only_deployment_selects_router_model_id():
"""A deployment priced solely via ``tiered_pricing`` (no flat
input/output cost) must resolve cost against its ``router_model_id``
entry, which holds the tiered table, instead of the shared backend alias
that has custom pricing fields stripped. Regression for tier-only models
(e.g. dashscope/qwen3.7-plus) being billed as free.
"""
from litellm import Router
from litellm.cost_calculator import _select_model_name_for_cost_calc
router = Router(
model_list=[
{
"model_name": "qwen-3.7-plus",
"litellm_params": {
"model": "dashscope/qwen3.7-plus",
"api_key": "sk-fake",
},
"model_info": {
"tiered_pricing": [
{
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.6e-06,
"range": [0, 256000],
},
],
},
},
]
)
router_model_id = router.model_list[0]["model_info"]["id"]
entry = litellm.model_cost[router_model_id]
assert entry.get("input_cost_per_token") is None
assert entry.get("tiered_pricing") is not None
# The stripped shared alias must not carry tiered pricing.
assert litellm.model_cost["dashscope/qwen3.7-plus"].get("tiered_pricing") is None
selected = _select_model_name_for_cost_calc(
model="dashscope/qwen3.7-plus",
completion_response=None,
custom_pricing=True,
custom_llm_provider="dashscope",
router_model_id=router_model_id,
)
assert selected is not None
assert router_model_id in selected
def test_tiered_pricing_only_deployment_completion_cost_is_nonzero():
"""End-to-end: a tier-only deployment must produce the tiered cost, not
$0. Mirrors the reported dashscope/qwen3.7-plus trace (12 prompt + 377
completion tokens).
"""
from litellm import Router
from litellm.types.utils import Choices, Message
router = Router(
model_list=[
{
"model_name": "qwen-3.7-plus",
"litellm_params": {
"model": "dashscope/qwen3.7-plus",
"api_key": "sk-fake",
},
"model_info": {
"tiered_pricing": [
{
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.6e-06,
"range": [0, 256000],
},
{
"input_cost_per_token": 1.2e-06,
"output_cost_per_token": 4.8e-06,
"range": [256000, 1000000],
},
],
},
},
]
)
router_model_id = router.model_list[0]["model_info"]["id"]
response = ModelResponse(
model="dashscope/qwen3.7-plus",
choices=[Choices(index=0, message=Message(role="assistant", content="hi"))],
usage=Usage(prompt_tokens=12, completion_tokens=377, total_tokens=389),
)
response._hidden_params = {"custom_llm_provider": "dashscope", "model_id": router_model_id}
cost = completion_cost(
completion_response=response,
model="dashscope/qwen3.7-plus",
custom_llm_provider="dashscope",
custom_pricing=True,
router_model_id=router_model_id,
)
expected = 12 * 4e-07 + 377 * 1.6e-06
assert cost == pytest.approx(expected)
assert cost > 0
def test_azure_realtime_cost_calculator():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")