diff --git a/litellm/llms/dashscope/rerank/transformation.py b/litellm/llms/dashscope/rerank/transformation.py index 629f3cf4af7..2bbfe42bc2a 100644 --- a/litellm/llms/dashscope/rerank/transformation.py +++ b/litellm/llms/dashscope/rerank/transformation.py @@ -22,7 +22,7 @@ as supported only for gte-rerank-v2 / qwen3-vl-rerank. Docs - https://help.aliyun.com/zh/model-studio/text-rerank-api """ -from typing import Any, Dict, List, Optional, Union +from typing import Any, Dict, List, Optional, Tuple, Union import httpx @@ -38,6 +38,7 @@ from litellm.types.rerank import ( RerankResponseMeta, RerankTokens, ) +from litellm.types.utils import ModelInfo from ..common_utils import DashScopeError @@ -239,3 +240,33 @@ class DashScopeRerankConfig(BaseRerankConfig): message=error_message, headers=headers, ) + + def calculate_rerank_cost( + self, + model: str, + custom_llm_provider: Optional[str] = None, + billed_units: Optional[RerankBilledUnits] = None, + model_info: Optional[ModelInfo] = None, + ) -> Tuple[float, float]: + """ + qwen3-rerank is billed per token. DashScope reports the count in + usage.total_tokens, surfaced here as billed_units["total_tokens"]; + set the per-token price via `input_cost_per_token` on the model. + + The base BaseRerankConfig prices per query (input_cost_per_query * + search_units), which DashScope does not report — hence this override. + """ + if ( + model_info is None + or "input_cost_per_token" not in model_info + or model_info["input_cost_per_token"] is None + or billed_units is None + ): + return 0.0, 0.0 + + total_tokens = billed_units.get("total_tokens") + if total_tokens is None: + return 0.0, 0.0 + + input_cost = model_info["input_cost_per_token"] * total_tokens + return input_cost, 0.0 diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ec16f19799b..624316fe120 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -10992,6 +10992,13 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "dashscope/qwen3-rerank": { + "input_cost_per_token": 1e-07, + "litellm_provider": "dashscope", + "mode": "rerank", + "output_cost_per_token": 0.0, + "source": "https://www.alibabacloud.com/help/en/model-studio/model-pricing" + }, "dashscope/qwen3-vl-235b-a22b-instruct": { "input_cost_per_token": 4e-07, "litellm_provider": "dashscope", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6e1c79c4e39..b24c9a89ca7 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -10992,6 +10992,13 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "dashscope/qwen3-rerank": { + "input_cost_per_token": 1e-07, + "litellm_provider": "dashscope", + "mode": "rerank", + "output_cost_per_token": 0.0, + "source": "https://www.alibabacloud.com/help/en/model-studio/model-pricing" + }, "dashscope/qwen3-vl-235b-a22b-instruct": { "input_cost_per_token": 4e-07, "litellm_provider": "dashscope", diff --git a/tests/test_litellm/llms/dashscope/test_dashscope_rerank_transformation.py b/tests/test_litellm/llms/dashscope/test_dashscope_rerank_transformation.py index 0e8d58b6530..84f794cb073 100644 --- a/tests/test_litellm/llms/dashscope/test_dashscope_rerank_transformation.py +++ b/tests/test_litellm/llms/dashscope/test_dashscope_rerank_transformation.py @@ -12,12 +12,14 @@ import pytest sys.path.insert(0, os.path.abspath("../../../../..")) +import litellm +from litellm.cost_calculator import rerank_cost from litellm.llms.dashscope.common_utils import DashScopeError from litellm.llms.dashscope.rerank.transformation import ( DEFAULT_RERANK_URL, DashScopeRerankConfig, ) -from litellm.types.rerank import RerankResponse +from litellm.types.rerank import RerankBilledUnits, RerankResponse class TestDashScopeRerankURL: @@ -314,6 +316,80 @@ class TestDashScopeRerankResponse: assert err.status_code == 500 +class TestDashScopeRerankCost: + def setup_method(self): + self.config = DashScopeRerankConfig() + + def test_cost_is_token_based(self): + # qwen3-rerank bills per token: total_tokens * input_cost_per_token. + prompt_cost, completion_cost = self.config.calculate_rerank_cost( + model="qwen3-rerank", + billed_units=RerankBilledUnits(total_tokens=1000), + model_info={"input_cost_per_token": 0.00000005}, # $0.05 / 1M tokens + ) + assert abs(prompt_cost - 0.00005) < 1e-10 # 1000 * 0.00000005 + assert completion_cost == 0.0 + + def test_cost_zero_when_total_tokens_missing(self): + prompt_cost, completion_cost = self.config.calculate_rerank_cost( + model="qwen3-rerank", + billed_units=RerankBilledUnits(), + model_info={"input_cost_per_token": 0.00000005}, + ) + assert (prompt_cost, completion_cost) == (0.0, 0.0) + + def test_cost_zero_when_model_info_missing(self): + prompt_cost, completion_cost = self.config.calculate_rerank_cost( + model="qwen3-rerank", + billed_units=RerankBilledUnits(total_tokens=1000), + model_info=None, + ) + assert (prompt_cost, completion_cost) == (0.0, 0.0) + + def test_cost_zero_when_price_unset(self): + # Model registered without input_cost_per_token -> no billing. + prompt_cost, completion_cost = self.config.calculate_rerank_cost( + model="qwen3-rerank", + billed_units=RerankBilledUnits(total_tokens=1000), + model_info={}, + ) + assert (prompt_cost, completion_cost) == (0.0, 0.0) + + +class TestDashScopeRerankPriceMap: + def _load_local_cost_map(self, monkeypatch): + # setattr auto-restores litellm.model_cost after the test, so mutating + # the global map here can't leak into other tests. + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + def test_qwen3_rerank_price_registered(self, monkeypatch): + # Pin the built-in price entry so cost tracking works without per-model + # config: $0.1 / 1M tokens == 1e-07 per token. + self._load_local_cost_map(monkeypatch) + + info = litellm.get_model_info( + model="qwen3-rerank", custom_llm_provider="dashscope" + ) + assert info["input_cost_per_token"] == 1e-07 + assert info["mode"] == "rerank" + assert info["litellm_provider"] == "dashscope" + + def test_qwen3_rerank_billed_end_to_end(self, monkeypatch): + # Full production cost path: rerank_cost -> get_model_info -> + # DashScopeRerankConfig.calculate_rerank_cost. Guards the seam the + # direct unit tests skip. + self._load_local_cost_map(monkeypatch) + + prompt_cost, completion_cost = rerank_cost( + model="qwen3-rerank", + custom_llm_provider="dashscope", + billed_units=RerankBilledUnits(total_tokens=1000), + ) + assert abs(prompt_cost - 1e-04) < 1e-12 # 1000 * 1e-07 + assert completion_cost == 0.0 + + class TestProviderConfigManagerDispatch: def test_dashscope_returns_rerank_config(self): import litellm