mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cost): bill off-peak rates for deployments that set only off_peak_pricing
Cost lookup selects the deployment-scoped cost map entry only when custom pricing is detected and the entry carries a base pricing field. A deployment whose model_info held nothing but off_peak_pricing failed both conditions, so its schedule was silently ignored and every request billed at the shared backend rate. use_custom_pricing_for_model now also treats deployment-scoped pricing fields in the metadata model_info as custom pricing, and the router inherits the backend model's built-in base token rates onto such an entry at registration, which also lets cache pricing inheritance apply. Regression tests cover the registration, the detection, and the costed request end to end.
This commit is contained in:
parent
1ba13fcc25
commit
b0751169eb
3 changed files with 157 additions and 2 deletions
|
|
@ -111,6 +111,7 @@ from litellm.types.mcp import MCPPostCallResponseObject
|
|||
from litellm.types.prompts.init_prompts import PromptSpec
|
||||
from litellm.types.rerank import RerankResponse
|
||||
from litellm.types.utils import (
|
||||
DEPLOYMENT_SCOPED_PRICING_FIELDS,
|
||||
CachingDetails,
|
||||
CallTypes,
|
||||
CostBreakdown,
|
||||
|
|
@ -255,6 +256,7 @@ _STANDARD_LOGGING_METADATA_KEYS: Final[frozenset[str]] = frozenset(StandardLoggi
|
|||
|
||||
# Cache custom pricing keys as frozenset for O(1) lookups instead of looping through 49 keys
|
||||
_CUSTOM_PRICING_KEYS: Final[frozenset[str]] = frozenset(CustomPricingLiteLLMParams.model_fields.keys())
|
||||
_MODEL_INFO_CUSTOM_PRICING_KEYS: Final[frozenset[str]] = _CUSTOM_PRICING_KEYS | DEPLOYMENT_SCOPED_PRICING_FIELDS
|
||||
|
||||
sentry_sdk_instance = None
|
||||
capture_exception = None
|
||||
|
|
@ -5030,7 +5032,9 @@ def use_custom_pricing_for_model(litellm_params: dict | None) -> bool:
|
|||
"""
|
||||
Check if the model uses custom pricing
|
||||
|
||||
Returns True if any of `SPECIAL_MODEL_INFO_PARAMS` are present in `litellm_params` or `model_info`
|
||||
Returns True if any custom pricing field is present in `litellm_params`, or if
|
||||
any custom pricing or deployment-scoped pricing field (such as
|
||||
``off_peak_pricing``) is present in the metadata ``model_info``
|
||||
"""
|
||||
if litellm_params is None:
|
||||
return False
|
||||
|
|
@ -5048,7 +5052,7 @@ def use_custom_pricing_for_model(litellm_params: dict | None) -> bool:
|
|||
model_info: dict = metadata.get("model_info", {}) or {}
|
||||
|
||||
if model_info:
|
||||
matching_keys = _CUSTOM_PRICING_KEYS & model_info.keys()
|
||||
matching_keys = _MODEL_INFO_CUSTOM_PRICING_KEYS & model_info.keys()
|
||||
for key in matching_keys:
|
||||
if model_info.get(key) is not None:
|
||||
return True
|
||||
|
|
|
|||
|
|
@ -8163,6 +8163,40 @@ class Router:
|
|||
if backend_value is not None:
|
||||
model_info[field] = backend_value
|
||||
|
||||
@staticmethod
|
||||
def _inherit_builtin_base_rates_for_off_peak(
|
||||
model_info: dict, # mutable-ok: cost-map entry filled in place
|
||||
backend_model: str,
|
||||
custom_llm_provider: str | None,
|
||||
) -> None:
|
||||
"""Fill missing base token rates on a deployment entry that only sets
|
||||
``off_peak_pricing``, from the backend model's built-in cost map entry.
|
||||
|
||||
Cost lookup selects the deployment-scoped entry over the shared backend
|
||||
entry only when the deployment entry carries a base pricing field, and
|
||||
``off_peak_pricing`` is deliberately kept off the shared entry, so a
|
||||
deployment spelling out only its off-peak schedule would otherwise
|
||||
never receive the discount. User-specified rates always win; no-op when
|
||||
any base pricing field is already set or the backend model has no
|
||||
canonical entry.
|
||||
"""
|
||||
if not model_info.get("off_peak_pricing"):
|
||||
return
|
||||
if any(
|
||||
model_info.get(field) is not None
|
||||
for field in ("input_cost_per_token", "input_cost_per_second", "tiered_pricing")
|
||||
):
|
||||
return
|
||||
try:
|
||||
backend_info: Final = litellm.get_model_info(model=backend_model, custom_llm_provider=custom_llm_provider)
|
||||
except Exception: # noqa: BLE001 # get_model_info raises plain Exception for an unmapped backend model
|
||||
return
|
||||
for field in ("input_cost_per_token", "output_cost_per_token"):
|
||||
if model_info.get(field) is None:
|
||||
backend_value = backend_info.get(field)
|
||||
if backend_value is not None:
|
||||
model_info[field] = backend_value
|
||||
|
||||
@staticmethod
|
||||
def _inherit_builtin_tiered_output_rate(
|
||||
model_info: dict, backend_model: str, custom_llm_provider: str | None
|
||||
|
|
@ -8251,6 +8285,11 @@ class Router:
|
|||
if deployment.litellm_params.get(field) is not None:
|
||||
_model_info[field] = deployment.litellm_params[field]
|
||||
|
||||
Router._inherit_builtin_base_rates_for_off_peak(
|
||||
model_info=_model_info,
|
||||
backend_model=deployment.litellm_params.model,
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
if _model_info.get("input_cost_per_token") is not None:
|
||||
Router._inherit_builtin_cache_pricing(
|
||||
model_info=_model_info,
|
||||
|
|
@ -8992,6 +9031,11 @@ class Router:
|
|||
if field_value is not None:
|
||||
_model_info_dict[field] = field_value
|
||||
|
||||
Router._inherit_builtin_base_rates_for_off_peak(
|
||||
model_info=_model_info_dict,
|
||||
backend_model=deployment.litellm_params.model,
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
if _model_info_dict.get("input_cost_per_token") is not None:
|
||||
Router._inherit_builtin_cache_pricing(
|
||||
model_info=_model_info_dict,
|
||||
|
|
@ -9246,6 +9290,11 @@ class Router:
|
|||
field_value = deployment.litellm_params.get(field)
|
||||
if field_value is not None:
|
||||
model_info[field] = field_value
|
||||
Router._inherit_builtin_base_rates_for_off_peak(
|
||||
model_info=model_info,
|
||||
backend_model=deployment.litellm_params.model,
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
if model_info.get("input_cost_per_token") is not None:
|
||||
Router._inherit_builtin_cache_pricing(
|
||||
model_info=model_info,
|
||||
|
|
|
|||
|
|
@ -891,3 +891,105 @@ def test_router_deployments_sharing_backend_keep_their_own_off_peak_pricing():
|
|||
litellm.model_cost.pop(deployment_id, None)
|
||||
_restore_model_cost_entries(original_entries)
|
||||
del router
|
||||
|
||||
|
||||
def test_router_off_peak_only_deployment_inherits_builtin_base_rates():
|
||||
"""A deployment that sets only ``off_peak_pricing`` on its model_info must
|
||||
still be costed from its deployment-scoped entry: the base token rates are
|
||||
inherited from the backend model's built-in cost map entry, since the
|
||||
shared backend key deliberately never carries the off-peak block.
|
||||
"""
|
||||
from litellm import Router
|
||||
|
||||
block = {
|
||||
"hours_utc": "00:00-00:00",
|
||||
"input_cost_per_token": 5e-05,
|
||||
"output_cost_per_token": 1e-04,
|
||||
}
|
||||
shared_keys = ["gpt-4o-mini", "openai/gpt-4o-mini"]
|
||||
deployment_id = "offpeak-only-dep-1"
|
||||
original_entries = _snapshot_model_cost_entries(shared_keys + [deployment_id])
|
||||
builtin_info = litellm.get_model_info(model="openai/gpt-4o-mini")
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "offpeak-only",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o-mini",
|
||||
"api_key": "fake-key-for-registration",
|
||||
},
|
||||
"model_info": {"id": deployment_id, "off_peak_pricing": dict(block)},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
try:
|
||||
entry = litellm.model_cost[deployment_id]
|
||||
assert entry["off_peak_pricing"] == block
|
||||
assert entry["input_cost_per_token"] is not None
|
||||
assert entry["input_cost_per_token"] == builtin_info["input_cost_per_token"]
|
||||
assert entry["output_cost_per_token"] == builtin_info["output_cost_per_token"]
|
||||
for shared_key in shared_keys:
|
||||
shared_entry = litellm.model_cost.get(shared_key) or {}
|
||||
assert not shared_entry.get("off_peak_pricing")
|
||||
finally:
|
||||
_restore_model_cost_entries(original_entries)
|
||||
del router
|
||||
|
||||
|
||||
def test_use_custom_pricing_for_model_sees_off_peak_only_model_info():
|
||||
from litellm.litellm_core_utils.litellm_logging import use_custom_pricing_for_model
|
||||
|
||||
block = {"hours_utc": "00:00-00:00", "input_cost_per_token": 5e-05}
|
||||
assert use_custom_pricing_for_model({"metadata": {"model_info": {"off_peak_pricing": block}}}) is True
|
||||
assert use_custom_pricing_for_model({"metadata": {"model_info": {"off_peak_pricing": None}}}) is False
|
||||
assert use_custom_pricing_for_model({"metadata": {"model_info": {"id": "some-id"}}}) is False
|
||||
|
||||
|
||||
def test_completion_cost_applies_off_peak_only_deployment_pricing():
|
||||
"""End to end through the cost calculator: with ``custom_pricing`` set and
|
||||
a ``router_model_id`` whose entry carries only an always-on off-peak block,
|
||||
the request bills at the block's rates rather than the shared backend rate.
|
||||
"""
|
||||
from litellm import Router
|
||||
from litellm.types.utils import ModelResponse, Usage
|
||||
|
||||
block = {
|
||||
"hours_utc": "00:00-00:00",
|
||||
"input_cost_per_token": 5e-05,
|
||||
"output_cost_per_token": 1e-04,
|
||||
}
|
||||
shared_keys = ["gpt-4o-mini", "openai/gpt-4o-mini"]
|
||||
deployment_id = "offpeak-only-dep-2"
|
||||
original_entries = _snapshot_model_cost_entries(shared_keys + [deployment_id])
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "offpeak-only",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o-mini",
|
||||
"api_key": "fake-key-for-registration",
|
||||
},
|
||||
"model_info": {"id": deployment_id, "off_peak_pricing": dict(block)},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
try:
|
||||
response = ModelResponse(
|
||||
model="gpt-4o-mini",
|
||||
usage=Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150),
|
||||
)
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model="openai/gpt-4o-mini",
|
||||
custom_llm_provider="openai",
|
||||
custom_pricing=True,
|
||||
router_model_id=deployment_id,
|
||||
)
|
||||
assert cost == pytest.approx(100 * 5e-05 + 50 * 1e-04)
|
||||
finally:
|
||||
_restore_model_cost_entries(original_entries)
|
||||
del router
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue