fix(model-registry): use proportional pricing for gpt-5.1-mini/nano; drop add_known_models side-effect

- gpt-5.1-mini and gpt-5.1-nano were priced identically to the base gpt-5.1 model.
  Apply the same ~30% (mini) and ~8% (nano) ratios used by the gpt-5.4 family:
    gpt-5.1-mini: input 3.75e-07, output 3e-06 (was 1.25e-06 / 1e-05)
    gpt-5.1-nano: input 1e-07,    output 8e-07 (was 1.25e-06 / 1e-05)
  Note: official OpenAI pricing for these aliases is not yet published; values
  are proportional estimates and will be updated when pricing is confirmed.
- Remove litellm.add_known_models() call from the Azure test autouse fixture —
  it mutates module-level sets (open_ai_chat_completion_models, etc.) that
  monkeypatch does not roll back, creating hidden test-ordering dependencies.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Mohamed EL-Habib 2026-05-23 18:25:31 +02:00
parent 8c0d74eab6
commit 3568dd1874
3 changed files with 24 additions and 25 deletions

View file

@ -20431,17 +20431,17 @@
"supports_minimal_reasoning_effort": true
},
"gpt-5.1-mini": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_priority": 2.5e-07,
"input_cost_per_token": 1.25e-06,
"input_cost_per_token_priority": 2.5e-06,
"cache_read_input_token_cost": 3.75e-08,
"cache_read_input_token_cost_priority": 7.5e-08,
"input_cost_per_token": 3.75e-07,
"input_cost_per_token_priority": 7.5e-07,
"litellm_provider": "openai",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 2e-05,
"output_cost_per_token": 3e-06,
"output_cost_per_token_priority": 6e-06,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
@ -20470,17 +20470,17 @@
"supports_minimal_reasoning_effort": true
},
"gpt-5.1-nano": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_priority": 2.5e-07,
"input_cost_per_token": 1.25e-06,
"input_cost_per_token_priority": 2.5e-06,
"cache_read_input_token_cost": 1e-08,
"cache_read_input_token_cost_priority": 2e-08,
"input_cost_per_token": 1e-07,
"input_cost_per_token_priority": 2e-07,
"litellm_provider": "openai",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 2e-05,
"output_cost_per_token": 8e-07,
"output_cost_per_token_priority": 1.6e-06,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"

View file

@ -20467,17 +20467,17 @@
"supports_minimal_reasoning_effort": true
},
"gpt-5.1-mini": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_priority": 2.5e-07,
"input_cost_per_token": 1.25e-06,
"input_cost_per_token_priority": 2.5e-06,
"cache_read_input_token_cost": 3.75e-08,
"cache_read_input_token_cost_priority": 7.5e-08,
"input_cost_per_token": 3.75e-07,
"input_cost_per_token_priority": 7.5e-07,
"litellm_provider": "openai",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 2e-05,
"output_cost_per_token": 3e-06,
"output_cost_per_token_priority": 6e-06,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
@ -20506,17 +20506,17 @@
"supports_minimal_reasoning_effort": true
},
"gpt-5.1-nano": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_priority": 2.5e-07,
"input_cost_per_token": 1.25e-06,
"input_cost_per_token_priority": 2.5e-06,
"cache_read_input_token_cost": 1e-08,
"cache_read_input_token_cost_priority": 2e-08,
"input_cost_per_token": 1e-07,
"input_cost_per_token_priority": 2e-07,
"litellm_provider": "openai",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 2e-05,
"output_cost_per_token": 8e-07,
"output_cost_per_token_priority": 1.6e-06,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"

View file

@ -16,7 +16,6 @@ def use_local_model_cost_map(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setattr(
litellm, "model_cost", get_model_cost_map(url=litellm.model_cost_map_url)
)
litellm.add_known_models(model_cost_map=litellm.model_cost)
def test_azure_gpt5_supports_reasoning_effort(config: AzureOpenAIGPT5Config):