mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Add above-512k pricing tier for MiniMax-M3 and correct its base rates (#30095)
* Add above-512k pricing tier support for MiniMax-M3 MiniMax-M3 doubles its per-token rates once a prompt exceeds 512k input tokens. The tiered cost parser already handles arbitrary thresholds, but get_model_info only copies whitelisted keys from ModelInfoBase, which had no 512k variants, so above_512k keys were silently dropped and long-context requests were priced at the flat rate. Add the input, output, and cache-read above_512k_tokens fields to ModelInfoBase and pass them through in get_model_info. Update the minimax/MiniMax-M3 entry with the tiered rates and correct the base rates, which matched the above-512k tier instead of the published base tier (https://platform.minimax.io/docs/guides/pricing-paygo). Fixes #29663. * Add above-512k keys to pricing schema, set MiniMax-M3 context to 1M Register the three new above_512k_tokens cost keys in the INTENDED_SCHEMA of test_aaamodel_prices_and_context_window_json_is_valid, declared the same way as the existing above_200k/above_272k tier keys, so the schema check accepts the MiniMax-M3 tiered pricing entry. Also raise MiniMax-M3 max_input_tokens from 512000 to 1000000 in both pricing JSONs. The MiniMax API docs (https://platform.minimax.io/docs/guides/text-generation) state the model supports a 1,000,000-token context window, and the pay-as-you-go pricing page (https://platform.minimax.io/docs/guides/pricing-paygo) prices input above 512k tokens, which only makes sense if inputs beyond 512k are accepted. This makes the above-512k pricing tier reachable.
This commit is contained in:
parent
b0754dbcb0
commit
48fb5f4e77
6 changed files with 68 additions and 8 deletions
|
|
@ -24392,9 +24392,12 @@
|
|||
"max_output_tokens": 8192
|
||||
},
|
||||
"minimax/MiniMax-M3": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token_above_512k_tokens": 6e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"output_cost_per_token_above_512k_tokens": 2.4e-06,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"cache_read_input_token_cost_above_512k_tokens": 1.2e-07,
|
||||
"litellm_provider": "minimax",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -24403,7 +24406,7 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_vision": true,
|
||||
"max_input_tokens": 512000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000
|
||||
},
|
||||
"mistral.devstral-2-123b": {
|
||||
|
|
|
|||
|
|
@ -197,6 +197,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
] # OpenAI priority service tier pricing
|
||||
cache_read_input_token_cost_above_200k_tokens: Optional[float]
|
||||
cache_read_input_token_cost_above_272k_tokens: Optional[float]
|
||||
cache_read_input_token_cost_above_512k_tokens: Optional[float]
|
||||
input_cost_per_character: Optional[float] # only for vertex ai models
|
||||
input_cost_per_audio_token: Optional[float]
|
||||
input_cost_per_token_above_128k_tokens: Optional[float] # only for vertex ai models
|
||||
|
|
@ -206,6 +207,9 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
input_cost_per_token_above_272k_tokens: Optional[
|
||||
float
|
||||
] # GPT-5.4/5.4-pro: prompts >272K priced at 2x input
|
||||
input_cost_per_token_above_512k_tokens: Optional[
|
||||
float
|
||||
] # MiniMax-M3: prompts >512K priced at 2x input
|
||||
input_cost_per_character_above_128k_tokens: Optional[
|
||||
float
|
||||
] # only for vertex ai models
|
||||
|
|
@ -239,6 +243,9 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
output_cost_per_token_above_272k_tokens: Optional[
|
||||
float
|
||||
] # GPT-5.4/5.4-pro: prompts >272K priced at 1.5x output
|
||||
output_cost_per_token_above_512k_tokens: Optional[
|
||||
float
|
||||
] # MiniMax-M3: prompts >512K priced at 2x output
|
||||
output_cost_per_character_above_128k_tokens: Optional[
|
||||
float
|
||||
] # only for vertex ai models
|
||||
|
|
|
|||
|
|
@ -5941,6 +5941,9 @@ def _get_model_info_helper( # noqa: PLR0915
|
|||
cache_read_input_token_cost_above_272k_tokens=_model_info.get(
|
||||
"cache_read_input_token_cost_above_272k_tokens", None
|
||||
),
|
||||
cache_read_input_token_cost_above_512k_tokens=_model_info.get(
|
||||
"cache_read_input_token_cost_above_512k_tokens", None
|
||||
),
|
||||
cache_read_input_token_cost_flex=_model_info.get(
|
||||
"cache_read_input_token_cost_flex", None
|
||||
),
|
||||
|
|
@ -5962,6 +5965,9 @@ def _get_model_info_helper( # noqa: PLR0915
|
|||
input_cost_per_token_above_272k_tokens=_model_info.get(
|
||||
"input_cost_per_token_above_272k_tokens", None
|
||||
),
|
||||
input_cost_per_token_above_512k_tokens=_model_info.get(
|
||||
"input_cost_per_token_above_512k_tokens", None
|
||||
),
|
||||
input_cost_per_query=_model_info.get("input_cost_per_query", None),
|
||||
input_cost_per_second=_model_info.get("input_cost_per_second", None),
|
||||
input_cost_per_audio_token=_model_info.get(
|
||||
|
|
@ -6017,6 +6023,9 @@ def _get_model_info_helper( # noqa: PLR0915
|
|||
output_cost_per_token_above_272k_tokens=_model_info.get(
|
||||
"output_cost_per_token_above_272k_tokens", None
|
||||
),
|
||||
output_cost_per_token_above_512k_tokens=_model_info.get(
|
||||
"output_cost_per_token_above_512k_tokens", None
|
||||
),
|
||||
output_cost_per_second=_model_info.get("output_cost_per_second", None),
|
||||
output_cost_per_second_1080p=_model_info.get(
|
||||
"output_cost_per_second_1080p", None
|
||||
|
|
|
|||
|
|
@ -24392,9 +24392,12 @@
|
|||
"max_output_tokens": 8192
|
||||
},
|
||||
"minimax/MiniMax-M3": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token_above_512k_tokens": 6e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"output_cost_per_token_above_512k_tokens": 2.4e-06,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"cache_read_input_token_cost_above_512k_tokens": 1.2e-07,
|
||||
"litellm_provider": "minimax",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -24403,7 +24406,7 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_vision": true,
|
||||
"max_input_tokens": 512000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000
|
||||
},
|
||||
"mistral.devstral-2-123b": {
|
||||
|
|
|
|||
|
|
@ -328,6 +328,41 @@ def test_generic_cost_per_token_gpt54_above_272k_tokens():
|
|||
assert round(completion_cost, 10) == round(expected_completion, 10)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_minimax_m3_above_512k_tokens():
|
||||
"""MiniMax-M3: prompts >512K input tokens priced at 2x input, output, and cache read."""
|
||||
model = "minimax/MiniMax-M3"
|
||||
custom_llm_provider = "minimax"
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
model_cost_map = litellm.model_cost[model]
|
||||
prompt_tokens = 600000
|
||||
cached_tokens = 100000
|
||||
completion_tokens = 1000
|
||||
usage = Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens),
|
||||
)
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
expected_prompt = (
|
||||
model_cost_map["input_cost_per_token_above_512k_tokens"]
|
||||
* (prompt_tokens - cached_tokens)
|
||||
+ model_cost_map["cache_read_input_token_cost_above_512k_tokens"]
|
||||
* cached_tokens
|
||||
)
|
||||
expected_completion = (
|
||||
model_cost_map["output_cost_per_token_above_512k_tokens"] * completion_tokens
|
||||
)
|
||||
assert round(prompt_cost, 10) == round(expected_prompt, 10)
|
||||
assert round(completion_cost, 10) == round(expected_completion, 10)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_gpt55():
|
||||
"""gpt-5.5: base pricing — $5/1M input, $30/1M output, $0.50/1M cached input."""
|
||||
model = "gpt-5.5"
|
||||
|
|
|
|||
|
|
@ -700,6 +700,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_read_input_token_cost": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_200k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_272k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_512k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": {
|
||||
"type": "number"
|
||||
|
|
@ -721,6 +722,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"input_cost_per_token_above_200k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_256k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_272k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_512k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_flex": {"type": "number"},
|
||||
"cache_read_input_token_cost_priority": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_200k_tokens_priority": {
|
||||
|
|
@ -811,6 +813,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"output_cost_per_token_above_200k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_256k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_272k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_512k_tokens": {"type": "number"},
|
||||
"output_cost_per_image_above_1024_and_1024_pixels": {"type": "number"},
|
||||
"output_cost_per_image_above_1024_and_1024_pixels_and_premium_image": {
|
||||
"type": "number"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue