mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(utils): preserve arbitrary above-threshold tiered pricing keys in get_model_info
get_model_info rebuilt ModelInfo by copying a fixed allow-list of input/output_cost_per_token_above_<N>_tokens keys (128k/200k/272k/512k), so any other threshold a user registered was dropped before reaching _get_token_base_cost, which already reads an arbitrary threshold out of the key name. Custom tiers such as above_500k_tokens were silently ignored and billing fell back to the base per-token rate. Carry over any _above_<N>_tokens cost key present on the source cost-map entry that the fixed fields miss Fixes #30344
This commit is contained in:
parent
e4a53f50de
commit
5424ccae29
2 changed files with 46 additions and 1 deletions
|
|
@ -5842,6 +5842,9 @@ def _is_potential_model_name_in_model_cost(
|
|||
)
|
||||
|
||||
|
||||
_ABOVE_THRESHOLD_COST_KEY = re.compile(r"_above_\d+k?_tokens$")
|
||||
|
||||
|
||||
def _get_model_info_helper(
|
||||
model: str,
|
||||
custom_llm_provider: Optional[str] = None,
|
||||
|
|
@ -6019,7 +6022,7 @@ def _get_model_info_helper(
|
|||
)
|
||||
_output_cost_per_token = 0
|
||||
|
||||
return ModelInfoBase(
|
||||
returned_model_info = ModelInfoBase(
|
||||
key=key,
|
||||
max_tokens=_model_info.get("max_tokens", None),
|
||||
max_input_tokens=_model_info.get("max_input_tokens", None),
|
||||
|
|
@ -6236,6 +6239,13 @@ def _get_model_info_helper(
|
|||
uses_embed_content=_model_info.get("uses_embed_content", None),
|
||||
supports_image_size=_model_info.get("supports_image_size", None),
|
||||
)
|
||||
for cost_key, cost_value in _model_info.items():
|
||||
if (
|
||||
cost_key not in returned_model_info
|
||||
and _ABOVE_THRESHOLD_COST_KEY.search(cost_key) is not None
|
||||
):
|
||||
returned_model_info[cost_key] = cost_value # type: ignore[literal-required]
|
||||
return returned_model_info
|
||||
except Exception as e:
|
||||
verbose_logger.debug(f"Error getting model info: {e}")
|
||||
raise Exception(
|
||||
|
|
|
|||
|
|
@ -384,6 +384,41 @@ def test_generic_cost_per_token_minimax_m3_above_512k_tokens():
|
|||
assert round(completion_cost, 10) == round(expected_completion, 10)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_honors_non_standard_above_threshold():
|
||||
"""Regression for #30344: get_model_info must keep arbitrary
|
||||
input/output_cost_per_token_above_<N>_tokens thresholds, not only the hard-coded
|
||||
128k/200k/272k/512k set, so a custom tier boundary is applied past its limit."""
|
||||
model = "litellm-test-non-standard-tier"
|
||||
custom_llm_provider = "openai"
|
||||
litellm.register_model(
|
||||
{
|
||||
model: {
|
||||
"litellm_provider": custom_llm_provider,
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1e-6,
|
||||
"output_cost_per_token": 2e-6,
|
||||
"input_cost_per_token_above_500k_tokens": 9e-6,
|
||||
"output_cost_per_token_above_500k_tokens": 18e-6,
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
prompt_tokens = 600000
|
||||
completion_tokens = 1000
|
||||
usage = Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
)
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
assert round(prompt_cost, 10) == round(9e-6 * prompt_tokens, 10)
|
||||
assert round(completion_cost, 10) == round(18e-6 * completion_tokens, 10)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_gpt55():
|
||||
"""gpt-5.5: base pricing — $5/1M input, $30/1M output, $0.50/1M cached input."""
|
||||
model = "gpt-5.5"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue