fix(cost): bill the fast service tier at the priority rate

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
milan 2026-07-31 04:04:04 +00:00 • committed by Devin AI
parent bf1a8fe403
commit 18b9e90d12
3 changed files with 79 additions and 7 deletions

View file

@ -2,7 +2,8 @@
## Helper utilities for cost_per_token()
from dataclasses import dataclass
from typing import Any, Literal, Optional, Tuple, TypedDict, cast
from types import MappingProxyType
from typing import Any, Literal, Mapping, Optional, Tuple, TypedDict, cast
import litellm
from litellm._logging import verbose_logger
@ -39,6 +40,14 @@ _VALID_DATA_RESIDENCIES = frozenset(r.value for r in DataResidency)
# of being rebuilt for every model_info key on every call.
_SERVICE_TIER_SUFFIXES: tuple[str, ...] = tuple(f"_{st.value}" for st in ServiceTier)
_SERVICE_TIER_TO_COST_KEY_SUFFIX: Mapping[str, str] = MappingProxyType(
{
ServiceTier.FLEX.value: ServiceTier.FLEX.value,
ServiceTier.PRIORITY.value: ServiceTier.PRIORITY.value,
ServiceTier.FAST.value: ServiceTier.PRIORITY.value,
}
)
def _get_token_detail_value(details: object, key: str) -> Optional[int]:
if isinstance(details, dict):
@ -177,7 +186,7 @@ def _get_service_tier_cost_key(base_key: str, service_tier: Optional[str]) -> st
Args:
base_key: The base cost key (e.g., "input_cost_per_token")
service_tier: The service tier ("flex", "priority", or None for standard)
service_tier: The service tier ("flex", "priority", "fast", or None for standard)
Returns:
str: The cost key to use (e.g., "input_cost_per_token_flex" or "input_cost_per_token")
@ -185,12 +194,11 @@ def _get_service_tier_cost_key(base_key: str, service_tier: Optional[str]) -> st
if service_tier is None:
return base_key
# Only use service tier specific keys for "flex" and "priority"
if service_tier.lower() in [ServiceTier.FLEX.value, ServiceTier.PRIORITY.value]:
return f"{base_key}_{service_tier.lower()}"
suffix = _SERVICE_TIER_TO_COST_KEY_SUFFIX.get(service_tier.lower())
if suffix is None:
return base_key
# For any other service tier, use standard pricing
return base_key
return f"{base_key}_{suffix}"
def _parse_above_token_threshold(key: str) -> float:

View file

@ -3847,6 +3847,7 @@ class ServiceTier(Enum):
AUTO = "auto"
FLEX = "flex"
PRIORITY = "priority"
FAST = "fast"
class DataResidency(Enum):

View file

@ -2447,3 +2447,66 @@ def test_generic_cost_per_token_gemini_35_flash_lite():
)
assert prompt_cost == pytest.approx(0.0003)
assert completion_cost == pytest.approx(0.00125)
def test_fast_service_tier_bills_at_the_priority_rate(_local_model_cost_map):
"""Regression: OpenAI's Fast mode replaced Priority Processing and costs 2x standard.
Before the fix "fast" fell through to standard pricing, so a Fast mode request
was billed at half of what it actually costs."""
from litellm.types.utils import Usage
usage = Usage(
prompt_tokens=1_000,
completion_tokens=500,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=200),
)
standard = generic_cost_per_token(
model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier=None
)
priority = generic_cost_per_token(
model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="priority"
)
fast = generic_cost_per_token(
model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="fast"
)
expected_prompt = 800 * 1e-05 + 200 * 1e-06
expected_completion = 500 * 6e-05
assert fast == priority
assert fast[0] == pytest.approx(expected_prompt, rel=1e-9)
assert fast[1] == pytest.approx(expected_completion, rel=1e-9)
assert fast[0] == pytest.approx(standard[0] * 2, rel=1e-9)
assert fast[1] == pytest.approx(standard[1] * 2, rel=1e-9)
def test_fast_service_tier_is_case_insensitive(_local_model_cost_map):
from litellm.types.utils import Usage
usage = Usage(prompt_tokens=1_000, completion_tokens=500)
assert generic_cost_per_token(
model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="FAST"
) == generic_cost_per_token(
model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="fast"
)
def test_fast_service_tier_matches_priority_above_the_context_threshold(_local_model_cost_map):
"""The above-threshold branch resolves its own cost keys, so the alias has to hold there too."""
from litellm.types.utils import Usage
usage = Usage(prompt_tokens=300_000, completion_tokens=1_000)
fast = generic_cost_per_token(
model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="fast"
)
priority = generic_cost_per_token(
model="gpt-5.6-sol", usage=usage, custom_llm_provider="openai", service_tier="priority"
)
assert fast == priority
assert fast[0] == pytest.approx(300_000 * 1e-05, rel=1e-9)
assert fast[1] == pytest.approx(1_000 * 4.5e-05, rel=1e-9)