From 18b9e90d123d581c3de238012acf44ad421c598d Mon Sep 17 00:00:00 2001 From: milan Date: Fri, 31 Jul 2026 04:04:04 +0000 Subject: [PATCH] fix(cost): bill the fast service tier at the priority rate Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../litellm_core_utils/llm_cost_calc/utils.py | 22 ++++--- litellm/types/utils.py | 1 + .../llm_cost_calc/test_llm_cost_calc_utils.py | 63 +++++++++++++++++++ 3 files changed, 79 insertions(+), 7 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 85ed0665ebf..fbc06b76c72 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -2,7 +2,8 @@ ## Helper utilities for cost_per_token() from dataclasses import dataclass -from typing import Any, Literal, Optional, Tuple, TypedDict, cast +from types import MappingProxyType +from typing import Any, Literal, Mapping, Optional, Tuple, TypedDict, cast import litellm from litellm._logging import verbose_logger @@ -39,6 +40,14 @@ _VALID_DATA_RESIDENCIES = frozenset(r.value for r in DataResidency) # of being rebuilt for every model_info key on every call. _SERVICE_TIER_SUFFIXES: tuple[str, ...] = tuple(f"_{st.value}" for st in ServiceTier) +_SERVICE_TIER_TO_COST_KEY_SUFFIX: Mapping[str, str] = MappingProxyType( + { + ServiceTier.FLEX.value: ServiceTier.FLEX.value, + ServiceTier.PRIORITY.value: ServiceTier.PRIORITY.value, + ServiceTier.FAST.value: ServiceTier.PRIORITY.value, + } +) + def _get_token_detail_value(details: object, key: str) -> Optional[int]: if isinstance(details, dict): @@ -177,7 +186,7 @@ def _get_service_tier_cost_key(base_key: str, service_tier: Optional[str]) -> st Args: base_key: The base cost key (e.g., "input_cost_per_token") - service_tier: The service tier ("flex", "priority", or None for standard) + service_tier: The service tier ("flex", "priority", "fast", or None for standard) Returns: str: The cost key to use (e.g., "input_cost_per_token_flex" or "input_cost_per_token") @@ -185,12 +194,11 @@ def _get_service_tier_cost_key(base_key: str, service_tier: Optional[str]) -> st if service_tier is None: return base_key - # Only use service tier specific keys for "flex" and "priority" - if service_tier.lower() in [ServiceTier.FLEX.value, ServiceTier.PRIORITY.value]: - return f"{base_key}_{service_tier.lower()}" + suffix = _SERVICE_TIER_TO_COST_KEY_SUFFIX.get(service_tier.lower()) + if suffix is None: + return base_key - # For any other service tier, use standard pricing - return base_key + return f"{base_key}_{suffix}" def _parse_above_token_threshold(key: str) -> float: diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 056592bbf93..18991f53e6f 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3847,6 +3847,7 @@ class ServiceTier(Enum): AUTO = "auto" FLEX = "flex" PRIORITY = "priority" + FAST = "fast" class DataResidency(Enum): diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index 3454f160cfa..866f8f71484 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -2447,3 +2447,66 @@ def test_generic_cost_per_token_gemini_35_flash_lite(): ) assert prompt_cost == pytest.approx(0.0003) assert completion_cost == pytest.approx(0.00125) + + +def test_fast_service_tier_bills_at_the_priority_rate(_local_model_cost_map): + """Regression: OpenAI's Fast mode replaced Priority Processing and costs 2x standard. + + Before the fix "fast" fell through to standard pricing, so a Fast mode request + was billed at half of what it actually costs.""" + from litellm.types.utils import Usage + + usage = Usage( + prompt_tokens=1_000, + completion_tokens=500, + prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=200), + ) + + standard = generic_cost_per_token( + model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier=None + ) + priority = generic_cost_per_token( + model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="priority" + ) + fast = generic_cost_per_token( + model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="fast" + ) + + expected_prompt = 800 * 1e-05 + 200 * 1e-06 + expected_completion = 500 * 6e-05 + + assert fast == priority + assert fast[0] == pytest.approx(expected_prompt, rel=1e-9) + assert fast[1] == pytest.approx(expected_completion, rel=1e-9) + assert fast[0] == pytest.approx(standard[0] * 2, rel=1e-9) + assert fast[1] == pytest.approx(standard[1] * 2, rel=1e-9) + + +def test_fast_service_tier_is_case_insensitive(_local_model_cost_map): + from litellm.types.utils import Usage + + usage = Usage(prompt_tokens=1_000, completion_tokens=500) + + assert generic_cost_per_token( + model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="FAST" + ) == generic_cost_per_token( + model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="fast" + ) + + +def test_fast_service_tier_matches_priority_above_the_context_threshold(_local_model_cost_map): + """The above-threshold branch resolves its own cost keys, so the alias has to hold there too.""" + from litellm.types.utils import Usage + + usage = Usage(prompt_tokens=300_000, completion_tokens=1_000) + + fast = generic_cost_per_token( + model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="fast" + ) + priority = generic_cost_per_token( + model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="priority" + ) + + assert fast == priority + assert fast[0] == pytest.approx(300_000 * 1e-05, rel=1e-9) + assert fast[1] == pytest.approx(1_000 * 4.5e-05, rel=1e-9)