fix(cost): treat a batch rate configured as zero as free, not unset

batch_cost_calculator gated the batch rate fields on truthiness, so a deployment
that configures input_cost_per_token_batches or its output twin as 0.0 was read
as having configured nothing and that token direction fell through to half the
standard rate. Layering declared rates over published ones made this reachable:
a deployment declaring only a zero batch rate previously kept a fabricated zero
on the standard field, which happened to bill nothing.

The two batch fields are now gated on presence. Verified no cost-map entry
changes behavior: the only three carrying a zero batch rate are embeddings, whose
standard output rate is also 0.0, so both paths yield the same zero.

Adds a parametrized regression over an explicit zero, an explicit non-zero, and
unset, plus coverage for the deployment id get_model_info cannot resolve, which
were the lines Codecov flagged.
This commit is contained in:
Marty Sullivan 2026-08-16 22:42:36 -04:00 • committed by mateo-berri
parent b593cef758
commit 800e1d4f35
3 changed files with 52 additions and 2 deletions

View file

@ -2160,7 +2160,7 @@ def batch_cost_calculator(
output_cost_per_token: Final = model_info.get("output_cost_per_token")
total_prompt_cost = 0.0
total_completion_cost = 0.0
if input_cost_per_token_batches:
if input_cost_per_token_batches is not None:
total_prompt_cost = usage.prompt_tokens * input_cost_per_token_batches
elif input_cost_per_token:
details: Final = parse_prompt_tokens_details(usage)
@ -2180,7 +2180,7 @@ def batch_cost_calculator(
cache_creation_cost: Final = model_info.get("cache_creation_input_token_cost") or input_cost_per_token
total_prompt_cost += cache_creation_tokens * cache_creation_cost / 2
if output_cost_per_token_batches:
if output_cost_per_token_batches is not None:
total_completion_cost = usage.completion_tokens * output_cost_per_token_batches
elif output_cost_per_token:
total_completion_cost = (

View file

@ -437,6 +437,19 @@ class TestGetRouterDeploymentModelInfo:
finally:
litellm.model_cost.pop(deployment_id, None)
def test_returns_none_when_the_deployment_id_resolves_no_provider(self, logging_obj) -> None:
"""A registration whose id get_model_info cannot resolve yields no pricing."""
deployment_id = "deploy-unresolvable-provider-1"
litellm.model_cost[deployment_id] = {"id": deployment_id, "input_cost_per_token": 4e-06}
logging_obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}}
logging_obj.model_call_details["model"] = None
logging_obj.model = None
try:
with patch.object(litellm, "get_model_info", side_effect=Exception("unresolvable")):
assert logging_obj.get_router_deployment_model_info() is None
finally:
litellm.model_cost.pop(deployment_id, None)
def test_falls_back_to_declared_rates_when_the_model_has_no_published_entry(self, logging_obj) -> None:
"""With no published entry to layer under, the declared rates still apply."""
deployment_id = "deploy-unpublished-model-1"

View file

@ -3595,6 +3595,43 @@ def test_batch_cost_calculator_cache_creation_falls_back_to_input_rate():
assert prompt_cost == pytest.approx((1000 * 3e-6 + 8000 * 3e-7 + 2000 * 3e-6) / 2)
@pytest.mark.parametrize(
"batch_rate,expected_prompt,expected_completion",
[
(0.0, 0.0, 0.0),
(1e-6, 1000 * 1e-6, 500 * 1e-6),
(None, 1000 * 3e-6 / 2, 500 * 15e-6 / 2),
],
ids=["explicit-zero", "explicit-nonzero", "unset"],
)
def test_batch_cost_calculator_honors_an_explicitly_zero_batch_rate(
batch_rate: float | None,
expected_prompt: float,
expected_completion: float,
) -> None:
"""A batch rate configured as 0.0 means free, not unset.
Gating the batch fields on truthiness read an explicit 0.0 as absent and
charged half the standard rate for that token direction instead.
"""
from litellm.cost_calculator import batch_cost_calculator
model_info: dict[str, float] = {"input_cost_per_token": 3e-6, "output_cost_per_token": 15e-6}
if batch_rate is not None:
model_info["input_cost_per_token_batches"] = batch_rate
model_info["output_cost_per_token_batches"] = batch_rate
prompt_cost, completion_cost_value = batch_cost_calculator(
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
model="claude-sonnet-4-5-20250929",
custom_llm_provider="anthropic",
model_info=model_info, # type: ignore[arg-type]
)
assert prompt_cost == pytest.approx(expected_prompt)
assert completion_cost_value == pytest.approx(expected_completion)
def test_combine_usage_objects_sums_mirrored_cache_write_fields_once():
"""
cache_write_tokens and cache_creation_tokens mirror each other on