fix(cost): bill off-peak rates for deployments that set only off_peak_pricing

Cost lookup selects the deployment-scoped cost map entry only when custom
pricing is detected and the entry carries a base pricing field. A deployment
whose model_info held nothing but off_peak_pricing failed both conditions, so
its schedule was silently ignored and every request billed at the shared
backend rate.

use_custom_pricing_for_model now also treats deployment-scoped pricing fields
in the metadata model_info as custom pricing, and the router inherits the
backend model's built-in base token rates onto such an entry at registration,
which also lets cache pricing inheritance apply. Regression tests cover the
registration, the detection, and the costed request end to end.
This commit is contained in:
mateo-berri 2026-09-01 12:14:46 -07:00
parent 1ba13fcc25
commit b0751169eb
3 changed files with 157 additions and 2 deletions

View file

@ -111,6 +111,7 @@ from litellm.types.mcp import MCPPostCallResponseObject
from litellm.types.prompts.init_prompts import PromptSpec
from litellm.types.rerank import RerankResponse
from litellm.types.utils import (
DEPLOYMENT_SCOPED_PRICING_FIELDS,
CachingDetails,
CallTypes,
CostBreakdown,
@ -255,6 +256,7 @@ _STANDARD_LOGGING_METADATA_KEYS: Final[frozenset[str]] = frozenset(StandardLoggi
# Cache custom pricing keys as frozenset for O(1) lookups instead of looping through 49 keys
_CUSTOM_PRICING_KEYS: Final[frozenset[str]] = frozenset(CustomPricingLiteLLMParams.model_fields.keys())
_MODEL_INFO_CUSTOM_PRICING_KEYS: Final[frozenset[str]] = _CUSTOM_PRICING_KEYS | DEPLOYMENT_SCOPED_PRICING_FIELDS
sentry_sdk_instance = None
capture_exception = None
@ -5030,7 +5032,9 @@ def use_custom_pricing_for_model(litellm_params: dict | None) -> bool:
"""
Check if the model uses custom pricing
Returns True if any of `SPECIAL_MODEL_INFO_PARAMS` are present in `litellm_params` or `model_info`
Returns True if any custom pricing field is present in `litellm_params`, or if
any custom pricing or deployment-scoped pricing field (such as
``off_peak_pricing``) is present in the metadata ``model_info``
"""
if litellm_params is None:
return False
@ -5048,7 +5052,7 @@ def use_custom_pricing_for_model(litellm_params: dict | None) -> bool:
model_info: dict = metadata.get("model_info", {}) or {}
if model_info:
matching_keys = _CUSTOM_PRICING_KEYS & model_info.keys()
matching_keys = _MODEL_INFO_CUSTOM_PRICING_KEYS & model_info.keys()
for key in matching_keys:
if model_info.get(key) is not None:
return True

View file

@ -8163,6 +8163,40 @@ class Router:
if backend_value is not None:
model_info[field] = backend_value
@staticmethod
def _inherit_builtin_base_rates_for_off_peak(
model_info: dict, # mutable-ok: cost-map entry filled in place
backend_model: str,
custom_llm_provider: str | None,
) -> None:
"""Fill missing base token rates on a deployment entry that only sets
``off_peak_pricing``, from the backend model's built-in cost map entry.
Cost lookup selects the deployment-scoped entry over the shared backend
entry only when the deployment entry carries a base pricing field, and
``off_peak_pricing`` is deliberately kept off the shared entry, so a
deployment spelling out only its off-peak schedule would otherwise
never receive the discount. User-specified rates always win; no-op when
any base pricing field is already set or the backend model has no
canonical entry.
"""
if not model_info.get("off_peak_pricing"):
return
if any(
model_info.get(field) is not None
for field in ("input_cost_per_token", "input_cost_per_second", "tiered_pricing")
):
return
try:
backend_info: Final = litellm.get_model_info(model=backend_model, custom_llm_provider=custom_llm_provider)
except Exception: # noqa: BLE001 # get_model_info raises plain Exception for an unmapped backend model
return
for field in ("input_cost_per_token", "output_cost_per_token"):
if model_info.get(field) is None:
backend_value = backend_info.get(field)
if backend_value is not None:
model_info[field] = backend_value
@staticmethod
def _inherit_builtin_tiered_output_rate(
model_info: dict, backend_model: str, custom_llm_provider: str | None
@ -8251,6 +8285,11 @@ class Router:
if deployment.litellm_params.get(field) is not None:
_model_info[field] = deployment.litellm_params[field]
Router._inherit_builtin_base_rates_for_off_peak(
model_info=_model_info,
backend_model=deployment.litellm_params.model,
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
)
if _model_info.get("input_cost_per_token") is not None:
Router._inherit_builtin_cache_pricing(
model_info=_model_info,
@ -8992,6 +9031,11 @@ class Router:
if field_value is not None:
_model_info_dict[field] = field_value
Router._inherit_builtin_base_rates_for_off_peak(
model_info=_model_info_dict,
backend_model=deployment.litellm_params.model,
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
)
if _model_info_dict.get("input_cost_per_token") is not None:
Router._inherit_builtin_cache_pricing(
model_info=_model_info_dict,
@ -9246,6 +9290,11 @@ class Router:
field_value = deployment.litellm_params.get(field)
if field_value is not None:
model_info[field] = field_value
Router._inherit_builtin_base_rates_for_off_peak(
model_info=model_info,
backend_model=deployment.litellm_params.model,
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
)
if model_info.get("input_cost_per_token") is not None:
Router._inherit_builtin_cache_pricing(
model_info=model_info,

View file

@ -891,3 +891,105 @@ def test_router_deployments_sharing_backend_keep_their_own_off_peak_pricing():
litellm.model_cost.pop(deployment_id, None)
_restore_model_cost_entries(original_entries)
del router
def test_router_off_peak_only_deployment_inherits_builtin_base_rates():
"""A deployment that sets only ``off_peak_pricing`` on its model_info must
still be costed from its deployment-scoped entry: the base token rates are
inherited from the backend model's built-in cost map entry, since the
shared backend key deliberately never carries the off-peak block.
"""
from litellm import Router
block = {
"hours_utc": "00:00-00:00",
"input_cost_per_token": 5e-05,
"output_cost_per_token": 1e-04,
}
shared_keys = ["gpt-4o-mini", "openai/gpt-4o-mini"]
deployment_id = "offpeak-only-dep-1"
original_entries = _snapshot_model_cost_entries(shared_keys + [deployment_id])
builtin_info = litellm.get_model_info(model="openai/gpt-4o-mini")
router = Router(
model_list=[
{
"model_name": "offpeak-only",
"litellm_params": {
"model": "openai/gpt-4o-mini",
"api_key": "fake-key-for-registration",
},
"model_info": {"id": deployment_id, "off_peak_pricing": dict(block)},
}
]
)
try:
entry = litellm.model_cost[deployment_id]
assert entry["off_peak_pricing"] == block
assert entry["input_cost_per_token"] is not None
assert entry["input_cost_per_token"] == builtin_info["input_cost_per_token"]
assert entry["output_cost_per_token"] == builtin_info["output_cost_per_token"]
for shared_key in shared_keys:
shared_entry = litellm.model_cost.get(shared_key) or {}
assert not shared_entry.get("off_peak_pricing")
finally:
_restore_model_cost_entries(original_entries)
del router
def test_use_custom_pricing_for_model_sees_off_peak_only_model_info():
from litellm.litellm_core_utils.litellm_logging import use_custom_pricing_for_model
block = {"hours_utc": "00:00-00:00", "input_cost_per_token": 5e-05}
assert use_custom_pricing_for_model({"metadata": {"model_info": {"off_peak_pricing": block}}}) is True
assert use_custom_pricing_for_model({"metadata": {"model_info": {"off_peak_pricing": None}}}) is False
assert use_custom_pricing_for_model({"metadata": {"model_info": {"id": "some-id"}}}) is False
def test_completion_cost_applies_off_peak_only_deployment_pricing():
"""End to end through the cost calculator: with ``custom_pricing`` set and
a ``router_model_id`` whose entry carries only an always-on off-peak block,
the request bills at the block's rates rather than the shared backend rate.
"""
from litellm import Router
from litellm.types.utils import ModelResponse, Usage
block = {
"hours_utc": "00:00-00:00",
"input_cost_per_token": 5e-05,
"output_cost_per_token": 1e-04,
}
shared_keys = ["gpt-4o-mini", "openai/gpt-4o-mini"]
deployment_id = "offpeak-only-dep-2"
original_entries = _snapshot_model_cost_entries(shared_keys + [deployment_id])
router = Router(
model_list=[
{
"model_name": "offpeak-only",
"litellm_params": {
"model": "openai/gpt-4o-mini",
"api_key": "fake-key-for-registration",
},
"model_info": {"id": deployment_id, "off_peak_pricing": dict(block)},
}
]
)
try:
response = ModelResponse(
model="gpt-4o-mini",
usage=Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150),
)
cost = litellm.completion_cost(
completion_response=response,
model="openai/gpt-4o-mini",
custom_llm_provider="openai",
custom_pricing=True,
router_model_id=deployment_id,
)
assert cost == pytest.approx(100 * 5e-05 + 50 * 1e-04)
finally:
_restore_model_cost_entries(original_entries)
del router