mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
fix(model-registry): use proportional pricing for gpt-5.1-mini/nano; drop add_known_models side-effect
- gpt-5.1-mini and gpt-5.1-nano were priced identically to the base gpt-5.1 model.
Apply the same ~30% (mini) and ~8% (nano) ratios used by the gpt-5.4 family:
gpt-5.1-mini: input 3.75e-07, output 3e-06 (was 1.25e-06 / 1e-05)
gpt-5.1-nano: input 1e-07, output 8e-07 (was 1.25e-06 / 1e-05)
Note: official OpenAI pricing for these aliases is not yet published; values
are proportional estimates and will be updated when pricing is confirmed.
- Remove litellm.add_known_models() call from the Azure test autouse fixture —
it mutates module-level sets (open_ai_chat_completion_models, etc.) that
monkeypatch does not roll back, creating hidden test-ordering dependencies.
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
parent
8c0d74eab6
commit
3568dd1874
3 changed files with 24 additions and 25 deletions
|
|
@ -20431,17 +20431,17 @@
|
|||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"gpt-5.1-mini": {
|
||||
"cache_read_input_token_cost": 1.25e-07,
|
||||
"cache_read_input_token_cost_priority": 2.5e-07,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"input_cost_per_token_priority": 2.5e-06,
|
||||
"cache_read_input_token_cost": 3.75e-08,
|
||||
"cache_read_input_token_cost_priority": 7.5e-08,
|
||||
"input_cost_per_token": 3.75e-07,
|
||||
"input_cost_per_token_priority": 7.5e-07,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_priority": 2e-05,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"output_cost_per_token_priority": 6e-06,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
|
|
@ -20470,17 +20470,17 @@
|
|||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"gpt-5.1-nano": {
|
||||
"cache_read_input_token_cost": 1.25e-07,
|
||||
"cache_read_input_token_cost_priority": 2.5e-07,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"input_cost_per_token_priority": 2.5e-06,
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"cache_read_input_token_cost_priority": 2e-08,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"input_cost_per_token_priority": 2e-07,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_priority": 2e-05,
|
||||
"output_cost_per_token": 8e-07,
|
||||
"output_cost_per_token_priority": 1.6e-06,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
|
|
|
|||
|
|
@ -20467,17 +20467,17 @@
|
|||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"gpt-5.1-mini": {
|
||||
"cache_read_input_token_cost": 1.25e-07,
|
||||
"cache_read_input_token_cost_priority": 2.5e-07,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"input_cost_per_token_priority": 2.5e-06,
|
||||
"cache_read_input_token_cost": 3.75e-08,
|
||||
"cache_read_input_token_cost_priority": 7.5e-08,
|
||||
"input_cost_per_token": 3.75e-07,
|
||||
"input_cost_per_token_priority": 7.5e-07,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_priority": 2e-05,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"output_cost_per_token_priority": 6e-06,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
|
|
@ -20506,17 +20506,17 @@
|
|||
"supports_minimal_reasoning_effort": true
|
||||
},
|
||||
"gpt-5.1-nano": {
|
||||
"cache_read_input_token_cost": 1.25e-07,
|
||||
"cache_read_input_token_cost_priority": 2.5e-07,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"input_cost_per_token_priority": 2.5e-06,
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"cache_read_input_token_cost_priority": 2e-08,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"input_cost_per_token_priority": 2e-07,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"output_cost_per_token_priority": 2e-05,
|
||||
"output_cost_per_token": 8e-07,
|
||||
"output_cost_per_token_priority": 1.6e-06,
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
|
|
|
|||
|
|
@ -16,7 +16,6 @@ def use_local_model_cost_map(monkeypatch: pytest.MonkeyPatch):
|
|||
monkeypatch.setattr(
|
||||
litellm, "model_cost", get_model_cost_map(url=litellm.model_cost_map_url)
|
||||
)
|
||||
litellm.add_known_models(model_cost_map=litellm.model_cost)
|
||||
|
||||
|
||||
def test_azure_gpt5_supports_reasoning_effort(config: AzureOpenAIGPT5Config):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue