fix(cost): inherit the backend's raw cost map entry for off-peak-only deployments

Filtering copied fields by name dropped companion billing rules like
web_search_billing_unit and the regional uplift multipliers, so grounding
and uplifts billed differently through the deployment entry. Copy the
backend's raw litellm.model_cost entry wholesale instead, which also
removes the synthesized-zero special case since the raw entry only holds
real values.
This commit is contained in:
mateo-berri 2026-09-01 13:05:25 -07:00
parent 0ec3e936b7
commit 60de5468d7
2 changed files with 41 additions and 15 deletions

View file

@ -8176,16 +8176,19 @@ class Router:
entry only when the deployment entry carries a base pricing field, and
``off_peak_pricing`` is deliberately kept off the shared entry, so a
deployment spelling out only its off-peak schedule would otherwise
never receive the discount. Every price-bearing backend field is
copied, not just the flat token rates: threshold, tiered, service-tier,
cache, character, and per-second rates all carry over, so peak-hour
billing through the deployment entry matches the shared backend entry
exactly. Values are deep-copied to keep the builtin entry isolated, and
a flat token rate ``get_model_info`` synthesized as zero for a backend
without one is rejected, like ``_inherit_builtin_tiered_output_rate``
does, so a tiered-only backend is never marked explicitly priced free.
User-specified rates always win; no-op when any base pricing field is
already set or the backend model has no canonical entry.
never receive the discount. The backend model's entire canonical cost
map entry is copied, field by field, so threshold, tiered,
service-tier, cache, character, and per-second rates as well as
companion billing fields like ``web_search_billing_unit`` and the
regional uplift multipliers all carry over, and peak-hour billing
through the deployment entry matches the shared backend entry exactly.
The raw ``litellm.model_cost`` entry is the copy source rather than
``get_model_info``'s view of it, since that view synthesizes zero flat
token rates for backends without one and storing those would mark a
tiered-only backend explicitly priced free. Values are deep-copied to
keep the builtin entry isolated. User-specified fields always win;
no-op when any base pricing field is already set or the backend model
has no canonical entry.
"""
if not model_info.get("off_peak_pricing"):
return
@ -8198,13 +8201,12 @@ class Router:
backend_info: Final = litellm.get_model_info(model=backend_model, custom_llm_provider=custom_llm_provider)
except Exception: # noqa: BLE001 # get_model_info raises plain Exception for an unmapped backend model
return
for field, backend_value in backend_info.items():
if "cost" not in field and field != "tiered_pricing":
continue
backend_entry: Final = litellm.model_cost.get(backend_info.get("key") or "")
if not isinstance(backend_entry, dict):
return
for field, backend_value in backend_entry.items():
if model_info.get(field) is not None or backend_value is None:
continue
if field in ("input_cost_per_token", "output_cost_per_token") and not backend_value:
continue
model_info[field] = copy.deepcopy(backend_value)
@staticmethod

View file

@ -595,6 +595,30 @@ def test_inherit_builtin_base_rates_for_off_peak_carries_threshold_rates():
)
def test_inherit_builtin_base_rates_for_off_peak_carries_companion_billing_fields():
"""Billing rules that are not literal cost rates, like the web search
billing unit, must ride along, or grounding and regional uplifts would
bill differently through the deployment entry than through the shared
backend entry.
"""
backend_model = "gemini-3-pro-image"
raw_entry = litellm.model_cost[backend_model]
assert raw_entry.get("web_search_billing_unit") is not None
model_info = {
"off_peak_pricing": {"hours_utc": "00:00-00:00", "input_cost_per_token": 5e-07},
}
Router._inherit_builtin_base_rates_for_off_peak(
model_info=model_info,
backend_model=backend_model,
custom_llm_provider=None,
)
assert model_info["web_search_billing_unit"] == raw_entry["web_search_billing_unit"]
assert model_info["input_cost_per_token"] == raw_entry["input_cost_per_token"]
def test_inherit_builtin_base_rates_for_off_peak_tiered_only_backend_stores_no_zero():
"""A tiered-only backend has no flat token rates; get_model_info synthesizes
zeros for them, and storing those would mark the deployment explicitly