From bc977b76dc3340f7a24baa40a04b6207a19b09e4 Mon Sep 17 00:00:00 2001 From: Marty Sullivan Date: Sun, 16 Aug 2026 23:41:07 -0400 Subject: [PATCH] fix(batches): own deployment pricing per token direction, not per field Filling each cost field independently let a published batch rate outrank a standard rate the deployment configured itself: a deployment declaring only input_cost_per_token had its batches billed at the model's published batch price rather than half its own rate. Measured on a model that publishes both, that billed $0.001500 where the deployment's own rate meant $0.000500. Declaring either rate for a direction now claims that whole direction, so nothing published can displace it, and a direction the deployment is silent on still inherits both published rates. --- litellm/litellm_core_utils/litellm_logging.py | 23 +++++++---- .../test_litellm_logging.py | 41 +++++++++++++++++++ 2 files changed, 57 insertions(+), 7 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index c22ac9eb4c9..96d9aa744fd 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -611,8 +611,11 @@ class Logging(LiteLLMLoggingBaseClass): decides that: the router registers an entry for every deployment, and get_model_info fills absent costs with 0, so asking it directly cannot tell "configured as free" apart from "no pricing configured". A deployment - may declare only one side of its pricing, so a rate it leaves unset keeps - the model's published value instead of billing as zero. + may declare only one side of its pricing, so the side it leaves out keeps + the model's published rates instead of billing as zero. Ownership is per + token direction: declaring either rate for a direction takes that whole + direction, so a published batch rate can never displace a standard rate + the deployment configured itself. """ model_id: Final = self.get_router_model_id() if model_id is None: @@ -629,13 +632,19 @@ class Logging(LiteLLMLoggingBaseClass): published: Final = self._published_model_info() if published is None: return merged - if registered.get("input_cost_per_token") is None: + declares_input: Final = ( + registered.get("input_cost_per_token") is not None + or registered.get("input_cost_per_token_batches") is not None + ) + declares_output: Final = ( + registered.get("output_cost_per_token") is not None + or registered.get("output_cost_per_token_batches") is not None + ) + if not declares_input: merged["input_cost_per_token"] = published.get("input_cost_per_token") - if registered.get("output_cost_per_token") is None: - merged["output_cost_per_token"] = published.get("output_cost_per_token") - if registered.get("input_cost_per_token_batches") is None: merged["input_cost_per_token_batches"] = published.get("input_cost_per_token_batches") - if registered.get("output_cost_per_token_batches") is None: + if not declares_output: + merged["output_cost_per_token"] = published.get("output_cost_per_token") merged["output_cost_per_token_batches"] = published.get("output_cost_per_token_batches") return merged diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index 180889fc36f..007617c309d 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -437,6 +437,47 @@ class TestGetRouterDeploymentModelInfo: finally: litellm.model_cost.pop(deployment_id, None) + def test_a_published_batch_rate_never_displaces_a_declared_standard_rate(self) -> None: + """Ownership is per token direction, not per field. + + Filling the batch field from the published entry let that rate win, so a + deployment configuring only its standard rate had batches billed at the + published batch price instead of half the rate it configured. + """ + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + + model = "ft:gpt-3.5-turbo" + published = litellm.get_model_info(model=model) + assert published["input_cost_per_token_batches"] is not None + + deployment_id = "deploy-standard-input-only-1" + litellm.model_cost[deployment_id] = { + "id": deployment_id, + "input_cost_per_token": 1e-06, + "litellm_provider": "openai", + "mode": "chat", + } + obj = LiteLLMLoggingObj( + model=model, + messages=[], + stream=False, + call_type="aretrieve_batch", + start_time=time.time(), + litellm_call_id="direction-ownership", + function_id="f", + ) + obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}, "model": model} + obj.model_call_details["model"] = model + try: + info = obj.get_router_deployment_model_info() + assert info is not None + assert info["input_cost_per_token"] == 1e-06 + assert info["input_cost_per_token_batches"] is None + assert info["output_cost_per_token"] == published["output_cost_per_token"] + assert info["output_cost_per_token_batches"] == published["output_cost_per_token_batches"] + finally: + litellm.model_cost.pop(deployment_id, None) + def test_keeps_declared_rates_when_no_model_is_resolvable(self, logging_obj) -> None: """With no model to look a published entry up by, the declared rates stand alone.""" deployment_id = "deploy-no-model-at-all-1"