feat(cost): support time-based off-peak pricing in cost calculation

Some providers charge different per-token rates depending on the time of
day. DeepSeek, for example, has historically discounted its chat and
reasoner models during an off-peak window (16:30-00:30 UTC). LiteLLM's
cost map only modeled static per-token pricing, so cost tracking could
not stay accurate for these providers.

This adds optional off-peak pricing to a model entry: input_cost_per_token_off_peak,
output_cost_per_token_off_peak, cache_read_input_token_cost_off_peak, and an
off_peak_hours_utc window expressed as "HH:MM-HH:MM" in UTC (the window may
wrap past midnight). When the current UTC time falls inside the window, the
cost calculator uses the off-peak rates and otherwise falls back to the
standard rates, so existing models are unaffected. The fields are also
accepted as custom pricing on a deployment, so they can be set from the
proxy config or the SDK.

The window check is a pure function that takes the current time as an
argument, which keeps the regression tests deterministic without patching
the clock.
This commit is contained in:
Srivatsa03 2026-06-30 12:10:48 -05:00
parent d447be15b9
commit 36f3eb3a77
4 changed files with 239 additions and 0 deletions

View file

@ -3,6 +3,7 @@
from collections.abc import Mapping
from dataclasses import dataclass
from datetime import datetime, timezone
from types import MappingProxyType
from typing import Any, Final, Literal, TypedDict, cast
@ -266,10 +267,75 @@ def _get_tiered_base_costs(model_info: ModelInfo, usage: Usage) -> tuple[float,
)
def _is_within_off_peak_window(off_peak_hours_utc: str | list[str], current_time: datetime | None = None) -> bool:
"""Return True if current_time (UTC, defaulting to now) falls inside any off-peak window.
off_peak_hours_utc is a "HH:MM-HH:MM" string in UTC, or a list of such strings for providers
with multiple daily windows (e.g. ["16:30-00:30", "04:00-06:00"]). A window may wrap past
midnight. The start is inclusive and the end is exclusive; malformed windows are ignored.
"""
if current_time is None:
current_time = datetime.now(timezone.utc)
now = current_time.time()
windows = [off_peak_hours_utc] if isinstance(off_peak_hours_utc, str) else off_peak_hours_utc
for window in windows:
try:
start_str, end_str = window.split("-")
start = datetime.strptime(start_str.strip(), "%H:%M").time()
end = datetime.strptime(end_str.strip(), "%H:%M").time()
except (ValueError, AttributeError):
continue
if start <= end:
if start <= now < end:
return True
elif now >= start or now < end:
return True
return False
def _coerce_off_peak_rate(value: object, default: float) -> float:
if isinstance(value, bool):
return default
if isinstance(value, (int, float)):
return float(value)
if isinstance(value, str):
try:
return float(value)
except ValueError:
return default
return default
def _apply_off_peak_pricing(
model_info: ModelInfo,
current_time: datetime | None,
prompt_base_cost: float,
completion_base_cost: float,
cache_read_cost: float,
) -> tuple[float, float, float]:
"""Swap in off-peak per-token rates when the current UTC time is inside one of the model's
off_peak_pricing windows. Applied after threshold pricing so the discount is honored rather
than overwritten when a model combines off-peak and above-threshold rates. Any rate left
unset in off_peak_pricing falls back to the standard rate.
"""
off_peak = model_info.get("off_peak_pricing")
if not off_peak:
return prompt_base_cost, completion_base_cost, cache_read_cost
hours_utc = off_peak.get("hours_utc")
if not hours_utc or not _is_within_off_peak_window(hours_utc, current_time):
return prompt_base_cost, completion_base_cost, cache_read_cost
return (
_coerce_off_peak_rate(off_peak.get("input_cost_per_token"), prompt_base_cost),
_coerce_off_peak_rate(off_peak.get("output_cost_per_token"), completion_base_cost),
_coerce_off_peak_rate(off_peak.get("cache_read_input_token_cost"), cache_read_cost),
)
def _get_token_base_cost(
model_info: ModelInfo,
usage: Usage,
service_tier: str | None = None,
current_time: datetime | None = None,
*,
threshold_is_inclusive: bool = False,
) -> tuple[float, float, float, float, float]:
@ -321,6 +387,9 @@ def _get_token_base_cost(
k for k in model_info if k.startswith("input_cost_per_token_above_") and not k.endswith(_SERVICE_TIER_SUFFIXES)
]
if not threshold_keys:
prompt_base_cost, completion_base_cost, cache_read_cost = _apply_off_peak_pricing(
model_info, current_time, prompt_base_cost, completion_base_cost, cache_read_cost
)
return (
prompt_base_cost,
completion_base_cost,
@ -427,6 +496,9 @@ def _get_token_base_cost(
except Exception:
continue
prompt_base_cost, completion_base_cost, cache_read_cost = _apply_off_peak_pricing(
model_info, current_time, prompt_base_cost, completion_base_cost, cache_read_cost
)
return (
prompt_base_cost,
completion_base_cost,

View file

@ -190,6 +190,19 @@ class AgenticLoopParams(TypedDict, total=False):
"""The LLM provider name (e.g., 'bedrock', 'anthropic')"""
class OffPeakPricing(TypedDict, total=False):
"""Time-windowed off-peak rates for providers that discount by time of day (e.g. DeepSeek).
hours_utc is a "HH:MM-HH:MM" string in UTC, or a list of them for multiple daily windows;
a window may wrap past midnight. Any rate left unset falls back to the standard rate.
"""
hours_utc: str | list[str]
input_cost_per_token: float
output_cost_per_token: float
cache_read_input_token_cost: float
class ModelInfoBase(ProviderSpecificModelInfo, total=False):
key: Required[str] # the key in litellm.model_cost which is returned
@ -222,6 +235,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
# Smallest prefix this model will actually cache, whatever caching mechanism its provider uses.
# Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT.
prompt_cache_min_tokens: int | None
off_peak_pricing: OffPeakPricing | None # time-windowed off-peak rates
input_cost_per_character: float | None # only for vertex ai models
input_cost_per_audio_token: float | None
input_cost_per_token_above_128k_tokens: float | None # only for vertex ai models

View file

@ -5658,6 +5658,7 @@ def _get_model_info_helper(
cache_creation_input_token_cost_above_1hr=_model_info.get(
"cache_creation_input_token_cost_above_1hr", None
),
off_peak_pricing=_model_info.get("off_peak_pricing", None),
input_cost_per_character=_model_info.get("input_cost_per_character", None),
input_cost_per_token_above_128k_tokens=_model_info.get("input_cost_per_token_above_128k_tokens", None),
input_cost_per_token_above_200k_tokens=_model_info.get("input_cost_per_token_above_200k_tokens", None),

View file

@ -31,6 +31,7 @@ from litellm.litellm_core_utils.llm_cost_calc.utils import (
TokenTypeCostBreakdown,
_calculate_input_cost,
_get_token_base_cost,
_is_within_off_peak_window,
calculate_cache_writing_cost,
generic_cost_per_token,
get_token_type_cost_breakdown,
@ -3517,3 +3518,154 @@ def test_generic_cost_per_token_grok_46_long_context(_local_model_cost_map):
)
assert prompt_cost == pytest.approx(200_000 * 4e-06 + 50_000 * 1e-06)
assert completion_cost == pytest.approx(1_000 * 1.2e-05)
def test_is_within_off_peak_window_same_day():
from datetime import datetime, timezone
window = "09:00-17:00"
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 12, 0, tzinfo=timezone.utc)) is True
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 8, 59, tzinfo=timezone.utc)) is False
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 9, 0, tzinfo=timezone.utc)) is True
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 17, 0, tzinfo=timezone.utc)) is False
def test_is_within_off_peak_window_wraps_midnight():
from datetime import datetime, timezone
window = "16:30-00:30"
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 18, 0, tzinfo=timezone.utc)) is True
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 0, 15, tzinfo=timezone.utc)) is True
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 16, 30, tzinfo=timezone.utc)) is True
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 0, 30, tzinfo=timezone.utc)) is False
assert _is_within_off_peak_window(window, datetime(2026, 1, 1, 12, 0, tzinfo=timezone.utc)) is False
def test_is_within_off_peak_window_multiple_windows():
from datetime import datetime, timezone
# Providers like DeepSeek V4 have more than one daily peak/off-peak window.
windows = ["01:00-05:00", "13:00-16:00"]
assert _is_within_off_peak_window(windows, datetime(2026, 1, 1, 3, 0, tzinfo=timezone.utc)) is True
assert _is_within_off_peak_window(windows, datetime(2026, 1, 1, 14, 30, tzinfo=timezone.utc)) is True
assert _is_within_off_peak_window(windows, datetime(2026, 1, 1, 9, 0, tzinfo=timezone.utc)) is False
# a malformed entry in the list is ignored, valid entries still match
assert _is_within_off_peak_window(["bad", "13:00-16:00"], datetime(2026, 1, 1, 14, 0, tzinfo=timezone.utc)) is True
assert _is_within_off_peak_window([], datetime(2026, 1, 1, 14, 0, tzinfo=timezone.utc)) is False
def test_is_within_off_peak_window_malformed_returns_false():
from datetime import datetime, timezone
now = datetime(2026, 1, 1, 18, 0, tzinfo=timezone.utc)
assert _is_within_off_peak_window("not-a-window", now) is False
assert _is_within_off_peak_window("16:30", now) is False
assert _is_within_off_peak_window("25:00-26:00", now) is False
def test_get_token_base_cost_applies_off_peak_pricing():
from datetime import datetime, timezone
from typing import cast
from litellm.types.utils import ModelInfo
model_info = cast(
ModelInfo,
{
"input_cost_per_token": 1e-6,
"output_cost_per_token": 2e-6,
"cache_read_input_token_cost": 1e-7,
"off_peak_pricing": {
"hours_utc": "16:30-00:30",
"input_cost_per_token": 5e-7,
"output_cost_per_token": 1e-6,
"cache_read_input_token_cost": 5e-8,
},
},
)
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
off_peak = _get_token_base_cost(model_info, usage, current_time=datetime(2026, 1, 1, 18, 0, tzinfo=timezone.utc))
assert off_peak[0] == 5e-7
assert off_peak[1] == 1e-6
assert off_peak[4] == 5e-8
peak = _get_token_base_cost(model_info, usage, current_time=datetime(2026, 1, 1, 12, 0, tzinfo=timezone.utc))
assert peak[0] == 1e-6
assert peak[1] == 2e-6
assert peak[4] == 1e-7
def test_get_token_base_cost_off_peak_falls_back_to_standard_when_unset():
from datetime import datetime, timezone
from typing import cast
from litellm.types.utils import ModelInfo
model_info = cast(
ModelInfo,
{
"input_cost_per_token": 1e-6,
"output_cost_per_token": 2e-6,
"off_peak_pricing": {"hours_utc": "16:30-00:30", "input_cost_per_token": 5e-7},
},
)
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
result = _get_token_base_cost(model_info, usage, current_time=datetime(2026, 1, 1, 18, 0, tzinfo=timezone.utc))
assert result[0] == 5e-7
assert result[1] == 2e-6
def test_get_token_base_cost_off_peak_wins_over_threshold():
from datetime import datetime, timezone
from typing import cast
from litellm.types.utils import ModelInfo
model_info = cast(
ModelInfo,
{
"input_cost_per_token": 1e-6,
"output_cost_per_token": 2e-6,
"input_cost_per_token_above_200k_tokens": 3e-6,
"output_cost_per_token_above_200k_tokens": 4e-6,
"off_peak_pricing": {
"hours_utc": "16:30-00:30",
"input_cost_per_token": 5e-7,
"output_cost_per_token": 1e-6,
},
},
)
usage = Usage(prompt_tokens=250000, completion_tokens=250000, total_tokens=500000)
off_peak = _get_token_base_cost(model_info, usage, current_time=datetime(2026, 1, 1, 18, 0, tzinfo=timezone.utc))
assert off_peak[0] == 5e-7
assert off_peak[1] == 1e-6
peak = _get_token_base_cost(model_info, usage, current_time=datetime(2026, 1, 1, 12, 0, tzinfo=timezone.utc))
assert peak[0] == 3e-6
assert peak[1] == 4e-6
def test_get_model_info_propagates_off_peak_fields():
model_name = "test-off-peak-model"
off_peak_pricing = {
"hours_utc": "16:30-00:30",
"input_cost_per_token": 5e-7,
"output_cost_per_token": 1e-6,
"cache_read_input_token_cost": 5e-8,
}
litellm.register_model(
{
model_name: {
"litellm_provider": "openai",
"mode": "chat",
"input_cost_per_token": 1e-6,
"output_cost_per_token": 2e-6,
"off_peak_pricing": off_peak_pricing,
}
}
)
info = litellm.get_model_info(model=model_name)
assert info["off_peak_pricing"] == off_peak_pricing