mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(cost): treat a batch rate configured as zero as free, not unset
batch_cost_calculator gated the batch rate fields on truthiness, so a deployment that configures input_cost_per_token_batches or its output twin as 0.0 was read as having configured nothing and that token direction fell through to half the standard rate. Layering declared rates over published ones made this reachable: a deployment declaring only a zero batch rate previously kept a fabricated zero on the standard field, which happened to bill nothing. The two batch fields are now gated on presence. Verified no cost-map entry changes behavior: the only three carrying a zero batch rate are embeddings, whose standard output rate is also 0.0, so both paths yield the same zero. Adds a parametrized regression over an explicit zero, an explicit non-zero, and unset, plus coverage for the deployment id get_model_info cannot resolve, which were the lines Codecov flagged.
This commit is contained in:
parent
b593cef758
commit
800e1d4f35
3 changed files with 52 additions and 2 deletions
|
|
@ -2160,7 +2160,7 @@ def batch_cost_calculator(
|
|||
output_cost_per_token: Final = model_info.get("output_cost_per_token")
|
||||
total_prompt_cost = 0.0
|
||||
total_completion_cost = 0.0
|
||||
if input_cost_per_token_batches:
|
||||
if input_cost_per_token_batches is not None:
|
||||
total_prompt_cost = usage.prompt_tokens * input_cost_per_token_batches
|
||||
elif input_cost_per_token:
|
||||
details: Final = parse_prompt_tokens_details(usage)
|
||||
|
|
@ -2180,7 +2180,7 @@ def batch_cost_calculator(
|
|||
|
||||
cache_creation_cost: Final = model_info.get("cache_creation_input_token_cost") or input_cost_per_token
|
||||
total_prompt_cost += cache_creation_tokens * cache_creation_cost / 2
|
||||
if output_cost_per_token_batches:
|
||||
if output_cost_per_token_batches is not None:
|
||||
total_completion_cost = usage.completion_tokens * output_cost_per_token_batches
|
||||
elif output_cost_per_token:
|
||||
total_completion_cost = (
|
||||
|
|
|
|||
|
|
@ -437,6 +437,19 @@ class TestGetRouterDeploymentModelInfo:
|
|||
finally:
|
||||
litellm.model_cost.pop(deployment_id, None)
|
||||
|
||||
def test_returns_none_when_the_deployment_id_resolves_no_provider(self, logging_obj) -> None:
|
||||
"""A registration whose id get_model_info cannot resolve yields no pricing."""
|
||||
deployment_id = "deploy-unresolvable-provider-1"
|
||||
litellm.model_cost[deployment_id] = {"id": deployment_id, "input_cost_per_token": 4e-06}
|
||||
logging_obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}}
|
||||
logging_obj.model_call_details["model"] = None
|
||||
logging_obj.model = None
|
||||
try:
|
||||
with patch.object(litellm, "get_model_info", side_effect=Exception("unresolvable")):
|
||||
assert logging_obj.get_router_deployment_model_info() is None
|
||||
finally:
|
||||
litellm.model_cost.pop(deployment_id, None)
|
||||
|
||||
def test_falls_back_to_declared_rates_when_the_model_has_no_published_entry(self, logging_obj) -> None:
|
||||
"""With no published entry to layer under, the declared rates still apply."""
|
||||
deployment_id = "deploy-unpublished-model-1"
|
||||
|
|
|
|||
|
|
@ -3595,6 +3595,43 @@ def test_batch_cost_calculator_cache_creation_falls_back_to_input_rate():
|
|||
assert prompt_cost == pytest.approx((1000 * 3e-6 + 8000 * 3e-7 + 2000 * 3e-6) / 2)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"batch_rate,expected_prompt,expected_completion",
|
||||
[
|
||||
(0.0, 0.0, 0.0),
|
||||
(1e-6, 1000 * 1e-6, 500 * 1e-6),
|
||||
(None, 1000 * 3e-6 / 2, 500 * 15e-6 / 2),
|
||||
],
|
||||
ids=["explicit-zero", "explicit-nonzero", "unset"],
|
||||
)
|
||||
def test_batch_cost_calculator_honors_an_explicitly_zero_batch_rate(
|
||||
batch_rate: float | None,
|
||||
expected_prompt: float,
|
||||
expected_completion: float,
|
||||
) -> None:
|
||||
"""A batch rate configured as 0.0 means free, not unset.
|
||||
|
||||
Gating the batch fields on truthiness read an explicit 0.0 as absent and
|
||||
charged half the standard rate for that token direction instead.
|
||||
"""
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
model_info: dict[str, float] = {"input_cost_per_token": 3e-6, "output_cost_per_token": 15e-6}
|
||||
if batch_rate is not None:
|
||||
model_info["input_cost_per_token_batches"] = batch_rate
|
||||
model_info["output_cost_per_token_batches"] = batch_rate
|
||||
|
||||
prompt_cost, completion_cost_value = batch_cost_calculator(
|
||||
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
|
||||
model="claude-sonnet-4-5-20250929",
|
||||
custom_llm_provider="anthropic",
|
||||
model_info=model_info, # type: ignore[arg-type]
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(expected_prompt)
|
||||
assert completion_cost_value == pytest.approx(expected_completion)
|
||||
|
||||
|
||||
def test_combine_usage_objects_sums_mirrored_cache_write_fields_once():
|
||||
"""
|
||||
cache_write_tokens and cache_creation_tokens mirror each other on
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue