fix(batches): own deployment pricing per token direction, not per field

Filling each cost field independently let a published batch rate outrank a
standard rate the deployment configured itself: a deployment declaring only
input_cost_per_token had its batches billed at the model's published batch price
rather than half its own rate. Measured on a model that publishes both, that
billed $0.001500 where the deployment's own rate meant $0.000500.

Declaring either rate for a direction now claims that whole direction, so nothing
published can displace it, and a direction the deployment is silent on still
inherits both published rates.
This commit is contained in:
Marty Sullivan 2026-08-16 23:41:07 -04:00 • committed by mateo-berri
parent e7c2ce8624
commit bc977b76dc
2 changed files with 57 additions and 7 deletions

View file

@ -611,8 +611,11 @@ class Logging(LiteLLMLoggingBaseClass):
decides that: the router registers an entry for every deployment, and
get_model_info fills absent costs with 0, so asking it directly cannot
tell "configured as free" apart from "no pricing configured". A deployment
may declare only one side of its pricing, so a rate it leaves unset keeps
the model's published value instead of billing as zero.
may declare only one side of its pricing, so the side it leaves out keeps
the model's published rates instead of billing as zero. Ownership is per
token direction: declaring either rate for a direction takes that whole
direction, so a published batch rate can never displace a standard rate
the deployment configured itself.
"""
model_id: Final = self.get_router_model_id()
if model_id is None:
@ -629,13 +632,19 @@ class Logging(LiteLLMLoggingBaseClass):
published: Final = self._published_model_info()
if published is None:
return merged
if registered.get("input_cost_per_token") is None:
declares_input: Final = (
registered.get("input_cost_per_token") is not None
or registered.get("input_cost_per_token_batches") is not None
)
declares_output: Final = (
registered.get("output_cost_per_token") is not None
or registered.get("output_cost_per_token_batches") is not None
)
if not declares_input:
merged["input_cost_per_token"] = published.get("input_cost_per_token")
if registered.get("output_cost_per_token") is None:
merged["output_cost_per_token"] = published.get("output_cost_per_token")
if registered.get("input_cost_per_token_batches") is None:
merged["input_cost_per_token_batches"] = published.get("input_cost_per_token_batches")
if registered.get("output_cost_per_token_batches") is None:
if not declares_output:
merged["output_cost_per_token"] = published.get("output_cost_per_token")
merged["output_cost_per_token_batches"] = published.get("output_cost_per_token_batches")
return merged

View file

@ -437,6 +437,47 @@ class TestGetRouterDeploymentModelInfo:
finally:
litellm.model_cost.pop(deployment_id, None)
def test_a_published_batch_rate_never_displaces_a_declared_standard_rate(self) -> None:
"""Ownership is per token direction, not per field.
Filling the batch field from the published entry let that rate win, so a
deployment configuring only its standard rate had batches billed at the
published batch price instead of half the rate it configured.
"""
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
model = "ft:gpt-3.5-turbo"
published = litellm.get_model_info(model=model)
assert published["input_cost_per_token_batches"] is not None
deployment_id = "deploy-standard-input-only-1"
litellm.model_cost[deployment_id] = {
"id": deployment_id,
"input_cost_per_token": 1e-06,
"litellm_provider": "openai",
"mode": "chat",
}
obj = LiteLLMLoggingObj(
model=model,
messages=[],
stream=False,
call_type="aretrieve_batch",
start_time=time.time(),
litellm_call_id="direction-ownership",
function_id="f",
)
obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}, "model": model}
obj.model_call_details["model"] = model
try:
info = obj.get_router_deployment_model_info()
assert info is not None
assert info["input_cost_per_token"] == 1e-06
assert info["input_cost_per_token_batches"] is None
assert info["output_cost_per_token"] == published["output_cost_per_token"]
assert info["output_cost_per_token_batches"] == published["output_cost_per_token_batches"]
finally:
litellm.model_cost.pop(deployment_id, None)
def test_keeps_declared_rates_when_no_model_is_resolvable(self, logging_obj) -> None:
"""With no model to look a published entry up by, the declared rates stand alone."""
deployment_id = "deploy-no-model-at-all-1"