fix(model-costs): apply the Sol promo cut to the gpt-5.6 alias

OpenAI's model page for gpt-5.6 serves the GPT-5.6 Sol page and states
that the gpt-5.6 alias routes requests to GPT-5.6 Sol, so the alias bills
at Sol's rates. The registry entry was left on the pre-cut rates while
gpt-5.6-sol took the cut, overbilling gpt-5.6 callers by 25 percent on
input and 50 percent on output.

All 23 cost fields on gpt-5.6 now match gpt-5.6-sol, and a regression
test pins the two entries together so they cannot drift again.
This commit is contained in:
mateo-berri 2026-08-21 17:37:44 -07:00
parent e6a6016e3e
commit 2ad2bec0f0
3 changed files with 64 additions and 46 deletions

View file

@ -26041,33 +26041,33 @@
"supports_minimal_reasoning_effort": true
},
"gpt-5.6": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_272k_tokens": 1.25e-05,
"cache_creation_input_token_cost_above_272k_tokens_flex": 6.25e-06,
"cache_creation_input_token_cost_flex": 3.125e-06,
"cache_creation_input_token_cost_priority": 1.25e-05,
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
"cache_read_input_token_cost_above_272k_tokens_flex": 5e-07,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 1e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_272k_tokens": 1e-05,
"input_cost_per_token_above_272k_tokens_flex": 5e-06,
"input_cost_per_token_batches": 2.5e-06,
"input_cost_per_token_flex": 2.5e-06,
"input_cost_per_token_priority": 1e-05,
"cache_creation_input_token_cost": 5e-06,
"cache_creation_input_token_cost_above_272k_tokens": 1e-05,
"cache_creation_input_token_cost_above_272k_tokens_flex": 5e-06,
"cache_creation_input_token_cost_flex": 2.5e-06,
"cache_creation_input_token_cost_priority": 1e-05,
"cache_read_input_token_cost": 4e-07,
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"input_cost_per_token_above_272k_tokens_flex": 4e-06,
"input_cost_per_token_batches": 2e-06,
"input_cost_per_token_flex": 2e-06,
"input_cost_per_token_priority": 8e-06,
"litellm_provider": "openai",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3e-05,
"output_cost_per_token_above_272k_tokens": 4.5e-05,
"output_cost_per_token_above_272k_tokens_flex": 2.25e-05,
"output_cost_per_token_batches": 1.5e-05,
"output_cost_per_token_flex": 1.5e-05,
"output_cost_per_token_priority": 6e-05,
"output_cost_per_token": 2e-05,
"output_cost_per_token_above_272k_tokens": 3e-05,
"output_cost_per_token_above_272k_tokens_flex": 1.5e-05,
"output_cost_per_token_batches": 1e-05,
"output_cost_per_token_flex": 1e-05,
"output_cost_per_token_priority": 4e-05,
"regional_processing_uplift_multiplier_eu": 1.1,
"regional_processing_uplift_multiplier_us": 1.1,
"search_context_cost_per_query": {

View file

@ -26041,33 +26041,33 @@
"supports_minimal_reasoning_effort": true
},
"gpt-5.6": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_272k_tokens": 1.25e-05,
"cache_creation_input_token_cost_above_272k_tokens_flex": 6.25e-06,
"cache_creation_input_token_cost_flex": 3.125e-06,
"cache_creation_input_token_cost_priority": 1.25e-05,
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
"cache_read_input_token_cost_above_272k_tokens_flex": 5e-07,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 1e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_272k_tokens": 1e-05,
"input_cost_per_token_above_272k_tokens_flex": 5e-06,
"input_cost_per_token_batches": 2.5e-06,
"input_cost_per_token_flex": 2.5e-06,
"input_cost_per_token_priority": 1e-05,
"cache_creation_input_token_cost": 5e-06,
"cache_creation_input_token_cost_above_272k_tokens": 1e-05,
"cache_creation_input_token_cost_above_272k_tokens_flex": 5e-06,
"cache_creation_input_token_cost_flex": 2.5e-06,
"cache_creation_input_token_cost_priority": 1e-05,
"cache_read_input_token_cost": 4e-07,
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"input_cost_per_token_above_272k_tokens_flex": 4e-06,
"input_cost_per_token_batches": 2e-06,
"input_cost_per_token_flex": 2e-06,
"input_cost_per_token_priority": 8e-06,
"litellm_provider": "openai",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3e-05,
"output_cost_per_token_above_272k_tokens": 4.5e-05,
"output_cost_per_token_above_272k_tokens_flex": 2.25e-05,
"output_cost_per_token_batches": 1.5e-05,
"output_cost_per_token_flex": 1.5e-05,
"output_cost_per_token_priority": 6e-05,
"output_cost_per_token": 2e-05,
"output_cost_per_token_above_272k_tokens": 3e-05,
"output_cost_per_token_above_272k_tokens_flex": 1.5e-05,
"output_cost_per_token_batches": 1e-05,
"output_cost_per_token_flex": 1e-05,
"output_cost_per_token_priority": 4e-05,
"regional_processing_uplift_multiplier_eu": 1.1,
"regional_processing_uplift_multiplier_us": 1.1,
"search_context_cost_per_query": {

View file

@ -913,7 +913,7 @@ def test_generic_cost_per_token_gpt55_pro():
@pytest.mark.parametrize(
"model,input_cost,output_cost,cache_read_cost,cache_write_cost",
[
("gpt-5.6", 5e-6, 3e-5, 5e-7, 6.25e-6),
("gpt-5.6", 4e-6, 2e-5, 4e-7, 5e-6),
("gpt-5.6-sol", 4e-6, 2e-5, 4e-7, 5e-6),
("gpt-5.6-terra", 2e-6, 1.2e-5, 2e-7, 2.5e-6),
("gpt-5.6-luna", 2e-7, 1.2e-6, 2e-8, 2.5e-7),
@ -965,10 +965,28 @@ def test_generic_cost_per_token_gpt56(
assert round(completion_cost, 10) == round(output_cost * completion_tokens, 10)
def test_gpt_5_6_alias_prices_match_sol():
"""Regression: the bare gpt-5.6 alias routes to GPT-5.6 Sol, so every cost field on
the two entries has to hold the same value. They drifted once before, when Sol took
its promotional cut and gpt-5.6 was left on the pre-cut rates, overbilling callers
who used the alias."""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
alias = litellm.model_cost["gpt-5.6"]
sol = litellm.model_cost["gpt-5.6-sol"]
cost_fields = sorted(field for field in sol if "cost" in field)
assert len(cost_fields) == 23
for field in cost_fields:
assert alias.get(field) == sol.get(field), field
@pytest.mark.parametrize(
"model,flex_long_input_cost,flex_long_output_cost",
[
("gpt-5.6", 5e-6, 2.25e-5),
("gpt-5.6", 4e-6, 1.5e-5),
("gpt-5.6-sol", 4e-6, 1.5e-5),
("gpt-5.6-terra", 2e-6, 9e-6),
("gpt-5.6-luna", 2e-7, 9e-7),