mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(cost): bill per-query priced rerank deployments from their router model id
This commit is contained in:
parent
af363fc29c
commit
4d2352d0b5
2 changed files with 42 additions and 0 deletions
|
|
@ -813,6 +813,7 @@ def _select_model_name_for_cost_calc(
|
|||
if (
|
||||
entry.get("input_cost_per_token") is not None
|
||||
or entry.get("input_cost_per_second") is not None
|
||||
or entry.get("input_cost_per_query") is not None
|
||||
or entry.get("tiered_pricing") is not None
|
||||
):
|
||||
return_model = router_model_id
|
||||
|
|
|
|||
|
|
@ -1213,6 +1213,47 @@ def test_tiered_pricing_only_deployment_completion_cost_is_nonzero():
|
|||
assert cost > 0
|
||||
|
||||
|
||||
def test_per_query_priced_rerank_deployment_completion_cost_is_nonzero():
|
||||
"""A rerank deployment priced only via ``input_cost_per_query`` must resolve
|
||||
cost against its ``router_model_id`` entry: the shared backend alias has
|
||||
custom pricing stripped, so pricing it there bills every search unit as $0.
|
||||
"""
|
||||
from litellm import Router
|
||||
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "semantic-ranker-default-004",
|
||||
"litellm_params": {
|
||||
"model": "vertex_ai/semantic-ranker-default-004",
|
||||
"vertex_project": "test-project",
|
||||
"vertex_location": "us-east5",
|
||||
},
|
||||
"model_info": {"input_cost_per_query": 0.001},
|
||||
},
|
||||
]
|
||||
)
|
||||
router_model_id: Final = router.model_list[0]["model_info"]["id"]
|
||||
assert litellm.model_cost["vertex_ai/semantic-ranker-default-004"].get("input_cost_per_query") is None
|
||||
|
||||
response: Final = RerankResponse(
|
||||
id="vertex_ai_rerank_test",
|
||||
results=[{"index": 3, "relevance_score": 0.48}],
|
||||
meta={"billed_units": {"search_units": 3}},
|
||||
)
|
||||
|
||||
cost: Final = completion_cost(
|
||||
completion_response=response,
|
||||
model="vertex_ai/semantic-ranker-default-004",
|
||||
custom_llm_provider="vertex_ai",
|
||||
call_type="arerank",
|
||||
custom_pricing=True,
|
||||
router_model_id=router_model_id,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(3 * 0.001)
|
||||
|
||||
|
||||
def test_azure_realtime_cost_calculator(_local_model_cost_map):
|
||||
|
||||
cost = handle_realtime_stream_cost_calculation(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue