mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(model-costs): apply the Sol promo cut to the gpt-5.6 alias
OpenAI's model page for gpt-5.6 serves the GPT-5.6 Sol page and states that the gpt-5.6 alias routes requests to GPT-5.6 Sol, so the alias bills at Sol's rates. The registry entry was left on the pre-cut rates while gpt-5.6-sol took the cut, overbilling gpt-5.6 callers by 25 percent on input and 50 percent on output. All 23 cost fields on gpt-5.6 now match gpt-5.6-sol, and a regression test pins the two entries together so they cannot drift again.
This commit is contained in:
parent
e6a6016e3e
commit
2ad2bec0f0
3 changed files with 64 additions and 46 deletions
|
|
@ -26041,33 +26041,33 @@
|
|||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"gpt-5.6": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex": 6.25e-06,
|
||||
"cache_creation_input_token_cost_flex": 3.125e-06,
|
||||
"cache_creation_input_token_cost_priority": 1.25e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 5e-07,
|
||||
"cache_read_input_token_cost_flex": 2.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1e-05,
|
||||
"input_cost_per_token_above_272k_tokens_flex": 5e-06,
|
||||
"input_cost_per_token_batches": 2.5e-06,
|
||||
"input_cost_per_token_flex": 2.5e-06,
|
||||
"input_cost_per_token_priority": 1e-05,
|
||||
"cache_creation_input_token_cost": 5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex": 5e-06,
|
||||
"cache_creation_input_token_cost_flex": 2.5e-06,
|
||||
"cache_creation_input_token_cost_priority": 1e-05,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 8e-06,
|
||||
"input_cost_per_token_above_272k_tokens_flex": 4e-06,
|
||||
"input_cost_per_token_batches": 2e-06,
|
||||
"input_cost_per_token_flex": 2e-06,
|
||||
"input_cost_per_token_priority": 8e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.5e-05,
|
||||
"output_cost_per_token_above_272k_tokens_flex": 2.25e-05,
|
||||
"output_cost_per_token_batches": 1.5e-05,
|
||||
"output_cost_per_token_flex": 1.5e-05,
|
||||
"output_cost_per_token_priority": 6e-05,
|
||||
"output_cost_per_token": 2e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 3e-05,
|
||||
"output_cost_per_token_above_272k_tokens_flex": 1.5e-05,
|
||||
"output_cost_per_token_batches": 1e-05,
|
||||
"output_cost_per_token_flex": 1e-05,
|
||||
"output_cost_per_token_priority": 4e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.1,
|
||||
"regional_processing_uplift_multiplier_us": 1.1,
|
||||
"search_context_cost_per_query": {
|
||||
|
|
|
|||
|
|
@ -26041,33 +26041,33 @@
|
|||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"gpt-5.6": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex": 6.25e-06,
|
||||
"cache_creation_input_token_cost_flex": 3.125e-06,
|
||||
"cache_creation_input_token_cost_priority": 1.25e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 5e-07,
|
||||
"cache_read_input_token_cost_flex": 2.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1e-05,
|
||||
"input_cost_per_token_above_272k_tokens_flex": 5e-06,
|
||||
"input_cost_per_token_batches": 2.5e-06,
|
||||
"input_cost_per_token_flex": 2.5e-06,
|
||||
"input_cost_per_token_priority": 1e-05,
|
||||
"cache_creation_input_token_cost": 5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex": 5e-06,
|
||||
"cache_creation_input_token_cost_flex": 2.5e-06,
|
||||
"cache_creation_input_token_cost_priority": 1e-05,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 8e-06,
|
||||
"input_cost_per_token_above_272k_tokens_flex": 4e-06,
|
||||
"input_cost_per_token_batches": 2e-06,
|
||||
"input_cost_per_token_flex": 2e-06,
|
||||
"input_cost_per_token_priority": 8e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.5e-05,
|
||||
"output_cost_per_token_above_272k_tokens_flex": 2.25e-05,
|
||||
"output_cost_per_token_batches": 1.5e-05,
|
||||
"output_cost_per_token_flex": 1.5e-05,
|
||||
"output_cost_per_token_priority": 6e-05,
|
||||
"output_cost_per_token": 2e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 3e-05,
|
||||
"output_cost_per_token_above_272k_tokens_flex": 1.5e-05,
|
||||
"output_cost_per_token_batches": 1e-05,
|
||||
"output_cost_per_token_flex": 1e-05,
|
||||
"output_cost_per_token_priority": 4e-05,
|
||||
"regional_processing_uplift_multiplier_eu": 1.1,
|
||||
"regional_processing_uplift_multiplier_us": 1.1,
|
||||
"search_context_cost_per_query": {
|
||||
|
|
|
|||
|
|
@ -913,7 +913,7 @@ def test_generic_cost_per_token_gpt55_pro():
|
|||
@pytest.mark.parametrize(
|
||||
"model,input_cost,output_cost,cache_read_cost,cache_write_cost",
|
||||
[
|
||||
("gpt-5.6", 5e-6, 3e-5, 5e-7, 6.25e-6),
|
||||
("gpt-5.6", 4e-6, 2e-5, 4e-7, 5e-6),
|
||||
("gpt-5.6-sol", 4e-6, 2e-5, 4e-7, 5e-6),
|
||||
("gpt-5.6-terra", 2e-6, 1.2e-5, 2e-7, 2.5e-6),
|
||||
("gpt-5.6-luna", 2e-7, 1.2e-6, 2e-8, 2.5e-7),
|
||||
|
|
@ -965,10 +965,28 @@ def test_generic_cost_per_token_gpt56(
|
|||
assert round(completion_cost, 10) == round(output_cost * completion_tokens, 10)
|
||||
|
||||
|
||||
def test_gpt_5_6_alias_prices_match_sol():
|
||||
"""Regression: the bare gpt-5.6 alias routes to GPT-5.6 Sol, so every cost field on
|
||||
the two entries has to hold the same value. They drifted once before, when Sol took
|
||||
its promotional cut and gpt-5.6 was left on the pre-cut rates, overbilling callers
|
||||
who used the alias."""
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
alias = litellm.model_cost["gpt-5.6"]
|
||||
sol = litellm.model_cost["gpt-5.6-sol"]
|
||||
|
||||
cost_fields = sorted(field for field in sol if "cost" in field)
|
||||
assert len(cost_fields) == 23
|
||||
|
||||
for field in cost_fields:
|
||||
assert alias.get(field) == sol.get(field), field
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,flex_long_input_cost,flex_long_output_cost",
|
||||
[
|
||||
("gpt-5.6", 5e-6, 2.25e-5),
|
||||
("gpt-5.6", 4e-6, 1.5e-5),
|
||||
("gpt-5.6-sol", 4e-6, 1.5e-5),
|
||||
("gpt-5.6-terra", 2e-6, 9e-6),
|
||||
("gpt-5.6-luna", 2e-7, 9e-7),
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue