mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(batches): own deployment pricing per token direction, not per field
Filling each cost field independently let a published batch rate outrank a standard rate the deployment configured itself: a deployment declaring only input_cost_per_token had its batches billed at the model's published batch price rather than half its own rate. Measured on a model that publishes both, that billed $0.001500 where the deployment's own rate meant $0.000500. Declaring either rate for a direction now claims that whole direction, so nothing published can displace it, and a direction the deployment is silent on still inherits both published rates.
This commit is contained in:
parent
e7c2ce8624
commit
bc977b76dc
2 changed files with 57 additions and 7 deletions
|
|
@ -611,8 +611,11 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
decides that: the router registers an entry for every deployment, and
|
||||
get_model_info fills absent costs with 0, so asking it directly cannot
|
||||
tell "configured as free" apart from "no pricing configured". A deployment
|
||||
may declare only one side of its pricing, so a rate it leaves unset keeps
|
||||
the model's published value instead of billing as zero.
|
||||
may declare only one side of its pricing, so the side it leaves out keeps
|
||||
the model's published rates instead of billing as zero. Ownership is per
|
||||
token direction: declaring either rate for a direction takes that whole
|
||||
direction, so a published batch rate can never displace a standard rate
|
||||
the deployment configured itself.
|
||||
"""
|
||||
model_id: Final = self.get_router_model_id()
|
||||
if model_id is None:
|
||||
|
|
@ -629,13 +632,19 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
published: Final = self._published_model_info()
|
||||
if published is None:
|
||||
return merged
|
||||
if registered.get("input_cost_per_token") is None:
|
||||
declares_input: Final = (
|
||||
registered.get("input_cost_per_token") is not None
|
||||
or registered.get("input_cost_per_token_batches") is not None
|
||||
)
|
||||
declares_output: Final = (
|
||||
registered.get("output_cost_per_token") is not None
|
||||
or registered.get("output_cost_per_token_batches") is not None
|
||||
)
|
||||
if not declares_input:
|
||||
merged["input_cost_per_token"] = published.get("input_cost_per_token")
|
||||
if registered.get("output_cost_per_token") is None:
|
||||
merged["output_cost_per_token"] = published.get("output_cost_per_token")
|
||||
if registered.get("input_cost_per_token_batches") is None:
|
||||
merged["input_cost_per_token_batches"] = published.get("input_cost_per_token_batches")
|
||||
if registered.get("output_cost_per_token_batches") is None:
|
||||
if not declares_output:
|
||||
merged["output_cost_per_token"] = published.get("output_cost_per_token")
|
||||
merged["output_cost_per_token_batches"] = published.get("output_cost_per_token_batches")
|
||||
return merged
|
||||
|
||||
|
|
|
|||
|
|
@ -437,6 +437,47 @@ class TestGetRouterDeploymentModelInfo:
|
|||
finally:
|
||||
litellm.model_cost.pop(deployment_id, None)
|
||||
|
||||
def test_a_published_batch_rate_never_displaces_a_declared_standard_rate(self) -> None:
|
||||
"""Ownership is per token direction, not per field.
|
||||
|
||||
Filling the batch field from the published entry let that rate win, so a
|
||||
deployment configuring only its standard rate had batches billed at the
|
||||
published batch price instead of half the rate it configured.
|
||||
"""
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
model = "ft:gpt-3.5-turbo"
|
||||
published = litellm.get_model_info(model=model)
|
||||
assert published["input_cost_per_token_batches"] is not None
|
||||
|
||||
deployment_id = "deploy-standard-input-only-1"
|
||||
litellm.model_cost[deployment_id] = {
|
||||
"id": deployment_id,
|
||||
"input_cost_per_token": 1e-06,
|
||||
"litellm_provider": "openai",
|
||||
"mode": "chat",
|
||||
}
|
||||
obj = LiteLLMLoggingObj(
|
||||
model=model,
|
||||
messages=[],
|
||||
stream=False,
|
||||
call_type="aretrieve_batch",
|
||||
start_time=time.time(),
|
||||
litellm_call_id="direction-ownership",
|
||||
function_id="f",
|
||||
)
|
||||
obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}, "model": model}
|
||||
obj.model_call_details["model"] = model
|
||||
try:
|
||||
info = obj.get_router_deployment_model_info()
|
||||
assert info is not None
|
||||
assert info["input_cost_per_token"] == 1e-06
|
||||
assert info["input_cost_per_token_batches"] is None
|
||||
assert info["output_cost_per_token"] == published["output_cost_per_token"]
|
||||
assert info["output_cost_per_token_batches"] == published["output_cost_per_token_batches"]
|
||||
finally:
|
||||
litellm.model_cost.pop(deployment_id, None)
|
||||
|
||||
def test_keeps_declared_rates_when_no_model_is_resolvable(self, logging_obj) -> None:
|
||||
"""With no model to look a published entry up by, the declared rates stand alone."""
|
||||
deployment_id = "deploy-no-model-at-all-1"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue