mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cost): add the gpt-5.5-pro batch long-context tier and ignore malformed batch tier keys
This commit is contained in:
parent
fce7868d90
commit
9819e4e0f8
6 changed files with 62 additions and 7 deletions
|
|
@ -3,7 +3,7 @@
|
|||
"limit": 13429
|
||||
},
|
||||
"reportArgumentType": {
|
||||
"limit": 2198
|
||||
"limit": 1913
|
||||
},
|
||||
"reportAssignmentType": {
|
||||
"limit": 319
|
||||
|
|
|
|||
|
|
@ -27,11 +27,11 @@ from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import
|
|||
from litellm.litellm_core_utils.llm_cost_calc.utils import (
|
||||
CostCalculatorUtils,
|
||||
_generic_cost_per_character,
|
||||
_get_batch_cost_rates,
|
||||
_get_regional_uplift_multiplier,
|
||||
_get_service_tier_cost_key,
|
||||
calculate_cost_component,
|
||||
generic_cost_per_token,
|
||||
get_batch_cost_rates,
|
||||
get_billable_input_tokens,
|
||||
get_token_type_cost_breakdown,
|
||||
parse_prompt_tokens_details,
|
||||
|
|
@ -2242,7 +2242,7 @@ def batch_cost_calculator(
|
|||
if not model_info:
|
||||
return 0.0, 0.0
|
||||
|
||||
input_cost_per_token_batches, output_cost_per_token_batches = _get_batch_cost_rates(
|
||||
input_cost_per_token_batches, output_cost_per_token_batches = get_batch_cost_rates(
|
||||
model_info, usage, custom_llm_provider
|
||||
)
|
||||
input_cost_per_token: Final = model_info.get("input_cost_per_token")
|
||||
|
|
|
|||
|
|
@ -62,6 +62,7 @@ _SERVICE_TIER_TO_COST_KEY_SUFFIX: Final[Mapping[str, str]] = MappingProxyType(
|
|||
|
||||
_INCLUSIVE_THRESHOLD_PROVIDERS: Final = frozenset({"xai"})
|
||||
_BATCH_KEY_SUFFIX: Final = "_batches"
|
||||
_BATCH_TIER_INPUT_KEY: Final = re.compile(r"^input_cost_per_token_above_\d+k?_tokens_batches$")
|
||||
_NON_STANDARD_THRESHOLD_SUFFIXES: Final = (*_SERVICE_TIER_SUFFIXES, _BATCH_KEY_SUFFIX)
|
||||
|
||||
|
||||
|
|
@ -252,14 +253,12 @@ def _batch_rate(model_info: ModelInfo, key: str) -> float | None:
|
|||
return value if isinstance(value, (int, float)) else None
|
||||
|
||||
|
||||
def _get_batch_cost_rates(
|
||||
def get_batch_cost_rates(
|
||||
model_info: ModelInfo, usage: Usage, custom_llm_provider: str | None
|
||||
) -> tuple[float | None, float | None]:
|
||||
inclusive: Final = _uses_inclusive_token_thresholds(custom_llm_provider)
|
||||
tier_input_keys: Final = tuple(
|
||||
key
|
||||
for key, value in model_info.items()
|
||||
if key.startswith("input_cost_per_token_above_") and key.endswith(_BATCH_KEY_SUFFIX) and value is not None
|
||||
key for key, value in model_info.items() if _BATCH_TIER_INPUT_KEY.match(key) and value is not None
|
||||
)
|
||||
crossed_input_key: Final = next(
|
||||
(
|
||||
|
|
|
|||
|
|
@ -30839,6 +30839,7 @@
|
|||
"input_cost_per_token_above_272k_tokens": 6e-05,
|
||||
"input_cost_per_token_flex": 1.5e-05,
|
||||
"input_cost_per_token_batches": 1.5e-05,
|
||||
"input_cost_per_token_above_272k_tokens_batches": 3e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -30848,6 +30849,7 @@
|
|||
"output_cost_per_token_above_272k_tokens": 0.00027,
|
||||
"output_cost_per_token_flex": 9e-05,
|
||||
"output_cost_per_token_batches": 9e-05,
|
||||
"output_cost_per_token_above_272k_tokens_batches": 0.000135,
|
||||
"regional_processing_uplift_multiplier_eu": 1.1,
|
||||
"regional_processing_uplift_multiplier_us": 1.1,
|
||||
"search_context_cost_per_query": {
|
||||
|
|
@ -30889,6 +30891,7 @@
|
|||
"input_cost_per_token_above_272k_tokens": 6e-05,
|
||||
"input_cost_per_token_flex": 1.5e-05,
|
||||
"input_cost_per_token_batches": 1.5e-05,
|
||||
"input_cost_per_token_above_272k_tokens_batches": 3e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -30898,6 +30901,7 @@
|
|||
"output_cost_per_token_above_272k_tokens": 0.00027,
|
||||
"output_cost_per_token_flex": 9e-05,
|
||||
"output_cost_per_token_batches": 9e-05,
|
||||
"output_cost_per_token_above_272k_tokens_batches": 0.000135,
|
||||
"regional_processing_uplift_multiplier_eu": 1.1,
|
||||
"regional_processing_uplift_multiplier_us": 1.1,
|
||||
"search_context_cost_per_query": {
|
||||
|
|
|
|||
|
|
@ -30839,6 +30839,7 @@
|
|||
"input_cost_per_token_above_272k_tokens": 6e-05,
|
||||
"input_cost_per_token_flex": 1.5e-05,
|
||||
"input_cost_per_token_batches": 1.5e-05,
|
||||
"input_cost_per_token_above_272k_tokens_batches": 3e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -30848,6 +30849,7 @@
|
|||
"output_cost_per_token_above_272k_tokens": 0.00027,
|
||||
"output_cost_per_token_flex": 9e-05,
|
||||
"output_cost_per_token_batches": 9e-05,
|
||||
"output_cost_per_token_above_272k_tokens_batches": 0.000135,
|
||||
"regional_processing_uplift_multiplier_eu": 1.1,
|
||||
"regional_processing_uplift_multiplier_us": 1.1,
|
||||
"search_context_cost_per_query": {
|
||||
|
|
@ -30889,6 +30891,7 @@
|
|||
"input_cost_per_token_above_272k_tokens": 6e-05,
|
||||
"input_cost_per_token_flex": 1.5e-05,
|
||||
"input_cost_per_token_batches": 1.5e-05,
|
||||
"input_cost_per_token_above_272k_tokens_batches": 3e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -30898,6 +30901,7 @@
|
|||
"output_cost_per_token_above_272k_tokens": 0.00027,
|
||||
"output_cost_per_token_flex": 9e-05,
|
||||
"output_cost_per_token_batches": 9e-05,
|
||||
"output_cost_per_token_above_272k_tokens_batches": 0.000135,
|
||||
"regional_processing_uplift_multiplier_eu": 1.1,
|
||||
"regional_processing_uplift_multiplier_us": 1.1,
|
||||
"search_context_cost_per_query": {
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import cast
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -4546,3 +4547,50 @@ def test_regular_path_never_bills_the_batch_tier_keys(_local_model_cost_map, mon
|
|||
|
||||
assert prompt_cost == pytest.approx(300_035 * 2e-6)
|
||||
assert completion_cost == pytest.approx(64 * 8e-6)
|
||||
|
||||
|
||||
def _lacks_half_price_batch_tier(entry: dict) -> bool:
|
||||
return any(
|
||||
entry.get(f"{side}_cost_per_token_above_272k_tokens_batches")
|
||||
!= entry[f"{side}_cost_per_token_above_272k_tokens"] / 2
|
||||
for side in ("input", "output")
|
||||
)
|
||||
|
||||
|
||||
def test_every_openai_entry_with_long_context_and_batch_rates_carries_the_batch_tier(_local_model_cost_map):
|
||||
offenders = [
|
||||
name
|
||||
for name, entry in litellm.model_cost.items()
|
||||
if isinstance(entry, dict)
|
||||
and entry.get("litellm_provider") == "openai"
|
||||
and entry.get("input_cost_per_token_above_272k_tokens") is not None
|
||||
and entry.get("input_cost_per_token_batches") is not None
|
||||
and _lacks_half_price_batch_tier(entry)
|
||||
]
|
||||
|
||||
assert offenders == []
|
||||
|
||||
|
||||
def test_batch_cost_calculator_ignores_malformed_batch_tier_keys():
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
usage = Usage(prompt_tokens=300_035, completion_tokens=64, total_tokens=300_099)
|
||||
model_info = cast(
|
||||
ModelInfo,
|
||||
{
|
||||
"input_cost_per_token": 2e-6,
|
||||
"output_cost_per_token": 8e-6,
|
||||
"input_cost_per_token_batches": 1e-6,
|
||||
"output_cost_per_token_batches": 4e-6,
|
||||
"input_cost_per_token_above_272k_tokens_batches": 2e-6,
|
||||
"output_cost_per_token_above_272k_tokens_batches": 6e-6,
|
||||
"input_cost_per_token_above_lots_tokens_batches": 1.0,
|
||||
},
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = batch_cost_calculator(
|
||||
usage=usage, model="gpt-5.4", custom_llm_provider="openai", model_info=model_info
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(300_035 * 2e-6)
|
||||
assert completion_cost == pytest.approx(64 * 6e-6)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue