mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-30 01:52:18 +00:00
fix(cost_calculator): bill chat per-second pricing once with a new cost_per_second field (#43614)
* feat(cost_calculator): add cost_per_second for chat per-second pricing Keep legacy input_cost_per_second and output_cost_per_second as aliases for chat, completion, embedding and responses. When both legacy fields are set, input_cost_per_second wins Move Bedrock commitment rows to cost_per_second so they bill once Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(cost_calculator): drop legacy per-second fields from chat paths Keep Azure chat token pricing generic and update inert Voxtral rates and SageMaker examples Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost_calculator): recognize output-only per-second rates Include output_cost_per_second when checking whether a deployment cost entry has pricing so output-only legacy aliases remain attached to the deployment during cost selection Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(pricing): cover cost_per_second and legacy per-second aliases through the proxy Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(cost_calculator): drop output_cost_per_second as a chat per-second alias Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(cost_calculator): restore output_cost_per_second as a chat per-second fallback Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost-map): keep input_cost_per_second on bedrock commitment rows for older clients Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
0fe4028cd9
commit
abc85c2651
29 changed files with 441 additions and 140 deletions
|
|
@ -24,7 +24,7 @@ model_list:
|
|||
- model_name: sagemaker-completion-model
|
||||
litellm_params:
|
||||
model: sagemaker/berri-benchmarking-Llama-2-70b-chat-hf-4
|
||||
input_cost_per_second: 0.000420
|
||||
cost_per_second: 0.000420
|
||||
- model_name: text-embedding-ada-002
|
||||
litellm_params:
|
||||
model: azure/azure-embedding-model
|
||||
|
|
|
|||
|
|
@ -125,6 +125,8 @@ pub struct ModelInfo {
|
|||
pub computer_use_input_cost_per_1k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub computer_use_output_cost_per_1k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cost_per_second: Option<f64>,
|
||||
/// Reasoning effort the provider applies when the request omits reasoning_effort. Gates whether a non-default temperature or the top_p/logprobs sampling params are accepted, which hold only when the effort resolves to 'none'.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub default_reasoning_effort: Option<ReasoningEffort>,
|
||||
|
|
|
|||
|
|
@ -351,19 +351,27 @@ def _per_second_pricing_cost(
|
|||
return None
|
||||
if _has_token_or_tiered_pricing(model_info) or not _bills_wall_clock_seconds(model_info):
|
||||
return None
|
||||
cost_per_second: Final = model_info.get("cost_per_second")
|
||||
input_cost_per_second: Final = model_info.get("input_cost_per_second")
|
||||
output_cost_per_second: Final = model_info.get("output_cost_per_second")
|
||||
if input_cost_per_second is None and output_cost_per_second is None:
|
||||
resolved_cost_per_second: Final = (
|
||||
cost_per_second
|
||||
if cost_per_second is not None
|
||||
else input_cost_per_second
|
||||
if input_cost_per_second is not None
|
||||
else output_cost_per_second
|
||||
)
|
||||
if resolved_cost_per_second is None:
|
||||
return None
|
||||
|
||||
seconds: Final = (response_time_ms or 0.0) / 1000
|
||||
verbose_logger.debug(
|
||||
"For model=%s - input_cost_per_second: %s; output_cost_per_second: %s; response time: %s",
|
||||
"For model=%s - cost_per_second: %s; response time: %s",
|
||||
model,
|
||||
input_cost_per_second,
|
||||
output_cost_per_second,
|
||||
resolved_cost_per_second,
|
||||
response_time_ms,
|
||||
)
|
||||
return (input_cost_per_second or 0.0) * seconds, (output_cost_per_second or 0.0) * seconds
|
||||
return resolved_cost_per_second * seconds, 0.0
|
||||
|
||||
|
||||
def cost_per_token(
|
||||
|
|
@ -790,7 +798,9 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None
|
|||
return value if isinstance(value, str) and value else None
|
||||
|
||||
|
||||
_NON_TOKEN_RATE_FIELDS: Final = frozenset({"input_cost_per_second", "input_cost_per_query", "tiered_pricing"})
|
||||
_NON_TOKEN_RATE_FIELDS: Final = frozenset(
|
||||
{"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"}
|
||||
)
|
||||
|
||||
|
||||
def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool:
|
||||
|
|
|
|||
|
|
@ -155,6 +155,7 @@ def get_litellm_params(
|
|||
allm_passthrough_route=None,
|
||||
preset_cache_key=None,
|
||||
no_log=None,
|
||||
cost_per_second: float | None = None,
|
||||
input_cost_per_second=None,
|
||||
input_cost_per_token=None,
|
||||
output_cost_per_token=None,
|
||||
|
|
@ -216,6 +217,7 @@ def get_litellm_params(
|
|||
"preset_cache_key": preset_cache_key,
|
||||
"no-log": no_log or kwargs.get("no-log"),
|
||||
"stream_response": {}, # litellm_call_id: ModelResponse Dict
|
||||
"cost_per_second": cost_per_second,
|
||||
"input_cost_per_token": input_cost_per_token,
|
||||
"input_cost_per_second": input_cost_per_second,
|
||||
"output_cost_per_token": output_cost_per_token,
|
||||
|
|
|
|||
|
|
@ -3,12 +3,8 @@ Helper util for handling azure openai-specific cost calculation
|
|||
- e.g.: prompt caching, audio tokens
|
||||
"""
|
||||
|
||||
from typing import Final
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
|
||||
def cost_per_token(
|
||||
|
|
@ -27,26 +23,6 @@ def cost_per_token(
|
|||
Returns:
|
||||
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
|
||||
"""
|
||||
## GET MODEL INFO
|
||||
model_info: Final = get_model_info(model=model, custom_llm_provider="azure")
|
||||
|
||||
## Speech / Audio cost calculation (cost per second for TTS models)
|
||||
if (
|
||||
"output_cost_per_second" in model_info
|
||||
and model_info["output_cost_per_second"] is not None
|
||||
and response_time_ms is not None
|
||||
):
|
||||
verbose_logger.debug(
|
||||
"For model=%s - output_cost_per_second: %s; response time: %s",
|
||||
model,
|
||||
model_info.get("output_cost_per_second"),
|
||||
response_time_ms,
|
||||
)
|
||||
## COST PER SECOND ##
|
||||
prompt_cost: Final = 0.0
|
||||
completion_cost: Final = model_info["output_cost_per_second"] * response_time_ms / 1000
|
||||
return prompt_cost, completion_cost
|
||||
|
||||
## Use generic cost calculator for all other cases
|
||||
## This properly handles: text tokens, audio tokens, cached tokens, reasoning tokens, etc.
|
||||
return generic_cost_per_token(
|
||||
|
|
|
|||
|
|
@ -5353,6 +5353,7 @@ def completion(
|
|||
### CUSTOM MODEL COST ###
|
||||
input_cost_per_token: Final = kwargs.get("input_cost_per_token", None)
|
||||
output_cost_per_token: Final = kwargs.get("output_cost_per_token", None)
|
||||
cost_per_second: Final = kwargs.get("cost_per_second", None)
|
||||
input_cost_per_second: Final = kwargs.get("input_cost_per_second", None)
|
||||
output_cost_per_second: Final = kwargs.get("output_cost_per_second", None)
|
||||
### CUSTOM PROMPT TEMPLATE ###
|
||||
|
|
@ -5514,8 +5515,11 @@ def completion(
|
|||
|
||||
### REGISTER CUSTOM MODEL PRICING -- IF GIVEN ###
|
||||
if (
|
||||
input_cost_per_token is not None and output_cost_per_token is not None
|
||||
) or input_cost_per_second is not None:
|
||||
(input_cost_per_token is not None and output_cost_per_token is not None)
|
||||
or input_cost_per_second is not None
|
||||
or output_cost_per_second is not None
|
||||
or cost_per_second is not None
|
||||
):
|
||||
_register_custom_pricing_for_request(
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
|
|
@ -5657,6 +5661,7 @@ def completion(
|
|||
proxy_server_request=proxy_server_request,
|
||||
preset_cache_key=preset_cache_key,
|
||||
no_log=no_log,
|
||||
cost_per_second=cost_per_second,
|
||||
input_cost_per_second=input_cost_per_second,
|
||||
input_cost_per_token=input_cost_per_token,
|
||||
output_cost_per_second=output_cost_per_second,
|
||||
|
|
@ -6354,7 +6359,9 @@ def embedding(
|
|||
### CUSTOM MODEL COST ###
|
||||
input_cost_per_token: Final = kwargs.get("input_cost_per_token", None)
|
||||
output_cost_per_token: Final = kwargs.get("output_cost_per_token", None)
|
||||
cost_per_second: Final = kwargs.get("cost_per_second", None)
|
||||
input_cost_per_second: Final = kwargs.get("input_cost_per_second", None)
|
||||
output_cost_per_second: Final = kwargs.get("output_cost_per_second", None)
|
||||
openai_params: Final = [
|
||||
"user",
|
||||
"dimensions",
|
||||
|
|
@ -6395,7 +6402,12 @@ def embedding(
|
|||
)
|
||||
|
||||
### REGISTER CUSTOM MODEL PRICING -- IF GIVEN ###
|
||||
if (input_cost_per_token is not None and output_cost_per_token is not None) or input_cost_per_second is not None:
|
||||
if (
|
||||
(input_cost_per_token is not None and output_cost_per_token is not None)
|
||||
or input_cost_per_second is not None
|
||||
or output_cost_per_second is not None
|
||||
or cost_per_second is not None
|
||||
):
|
||||
_register_custom_pricing_for_request(
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
|
|
|
|||
|
|
@ -12682,43 +12682,43 @@
|
|||
"source": "https://developers.openai.com/api/docs/pricing"
|
||||
},
|
||||
"bedrock/*/1-month-commitment/cohere.command-light-text-v14": {
|
||||
"cost_per_second": 0.001902,
|
||||
"input_cost_per_second": 0.001902,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.001902,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/*/1-month-commitment/cohere.command-text-v14": {
|
||||
"cost_per_second": 0.011,
|
||||
"input_cost_per_second": 0.011,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.011,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/*/6-month-commitment/cohere.command-light-text-v14": {
|
||||
"cost_per_second": 0.0011416,
|
||||
"input_cost_per_second": 0.0011416,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0011416,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/*/6-month-commitment/cohere.command-text-v14": {
|
||||
"cost_per_second": 0.0066027,
|
||||
"input_cost_per_second": 0.0066027,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0066027,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/guardrails": {
|
||||
|
|
@ -12737,61 +12737,61 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.01475,
|
||||
"input_cost_per_second": 0.01475,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.01475,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.0455,
|
||||
"input_cost_per_second": 0.0455,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0455
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.0455,
|
||||
"input_cost_per_second": 0.0455,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0455,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.008194,
|
||||
"input_cost_per_second": 0.008194,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.008194,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.02527,
|
||||
"input_cost_per_second": 0.02527,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.02527
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.02527,
|
||||
"input_cost_per_second": 0.02527,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.02527,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/anthropic.claude-instant-v1": {
|
||||
|
|
@ -13241,61 +13241,61 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.01635,
|
||||
"input_cost_per_second": 0.01635,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.01635,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.0415,
|
||||
"input_cost_per_second": 0.0415,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0415
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.0415,
|
||||
"input_cost_per_second": 0.0415,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0415,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.009083,
|
||||
"input_cost_per_second": 0.009083,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.009083,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.02305,
|
||||
"input_cost_per_second": 0.02305,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.02305
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.02305,
|
||||
"input_cost_per_second": 0.02305,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.02305,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/anthropic.claude-instant-v1": {
|
||||
|
|
@ -13737,61 +13737,61 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.011,
|
||||
"input_cost_per_second": 0.011,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.011,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.0175,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0175
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.0175,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0175,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.00611,
|
||||
"input_cost_per_second": 0.00611,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00611,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.00972,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00972
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.00972,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00972,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/anthropic.claude-instant-v1": {
|
||||
|
|
@ -14385,61 +14385,61 @@
|
|||
"output_cost_per_token": 6e-07
|
||||
},
|
||||
"bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.011,
|
||||
"input_cost_per_second": 0.011,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.011,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.0175,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0175
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.0175,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0175,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.00611,
|
||||
"input_cost_per_second": 0.00611,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00611,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.00972,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00972
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.00972,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00972,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/anthropic.claude-instant-v1": {
|
||||
|
|
@ -38527,7 +38527,6 @@
|
|||
},
|
||||
"mistral/voxtral-small-2507": {
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"input_cost_per_second": 6.666666666666667e-05,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 32768,
|
||||
|
|
@ -38543,7 +38542,6 @@
|
|||
},
|
||||
"mistral/voxtral-small-latest": {
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"input_cost_per_second": 6.666666666666667e-05,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 32768,
|
||||
|
|
|
|||
|
|
@ -8801,7 +8801,7 @@ class Router:
|
|||
return
|
||||
if any(
|
||||
model_info.get(field) is not None
|
||||
for field in ("input_cost_per_token", "input_cost_per_second", "tiered_pricing")
|
||||
for field in ("input_cost_per_token", "input_cost_per_second", "cost_per_second", "tiered_pricing")
|
||||
):
|
||||
return
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -606,6 +606,7 @@ class LiteLLMParamsTypedDict(TypedDict, total=False):
|
|||
## CUSTOM PRICING ##
|
||||
input_cost_per_token: float | None
|
||||
output_cost_per_token: float | None
|
||||
cost_per_second: ReadOnly[float | None]
|
||||
input_cost_per_second: float | None
|
||||
output_cost_per_second: float | None
|
||||
output_cost_per_second_480p: ReadOnly[float | None]
|
||||
|
|
|
|||
|
|
@ -329,6 +329,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
input_cost_per_video_per_second: float | None # only for vertex ai models
|
||||
input_cost_per_audio_token_batches: ReadOnly[float | None]
|
||||
input_cost_per_image_token_batches: ReadOnly[float | None]
|
||||
cost_per_second: ReadOnly[float | None]
|
||||
input_cost_per_second: float | None # for OpenAI Speech models
|
||||
input_cost_per_token_batches: float | None
|
||||
input_cost_per_video_token_batches: ReadOnly[float | None]
|
||||
|
|
@ -2784,6 +2785,7 @@ class LoggedLiteLLMParams(TypedDict, total=False):
|
|||
acompletion: bool | None
|
||||
preset_cache_key: str | None
|
||||
no_log: bool | None
|
||||
cost_per_second: ReadOnly[float | None]
|
||||
input_cost_per_second: float | None
|
||||
input_cost_per_token: float | None
|
||||
output_cost_per_token: float | None
|
||||
|
|
@ -3709,6 +3711,7 @@ class MirroredPricingParams(BaseModel):
|
|||
|
||||
class CustomPricingLiteLLMParams(MirroredPricingParams):
|
||||
## CUSTOM PRICING ##
|
||||
cost_per_second: float | None = None
|
||||
input_cost_per_second: float | None = None
|
||||
output_cost_per_second: float | None = None
|
||||
output_cost_per_second_1080p: float | None = None
|
||||
|
|
|
|||
|
|
@ -6168,6 +6168,7 @@ def _get_model_info_helper(
|
|||
),
|
||||
input_cost_per_token_above_512k_tokens=_model_info.get("input_cost_per_token_above_512k_tokens", None),
|
||||
input_cost_per_query=_model_info.get("input_cost_per_query", None),
|
||||
cost_per_second=_model_info.get("cost_per_second", None),
|
||||
input_cost_per_second=_model_info.get("input_cost_per_second", None),
|
||||
input_cost_per_audio_token=_model_info.get("input_cost_per_audio_token", None),
|
||||
input_cost_per_image_token=_model_info.get("input_cost_per_image_token", None),
|
||||
|
|
|
|||
|
|
@ -12682,43 +12682,43 @@
|
|||
"source": "https://developers.openai.com/api/docs/pricing"
|
||||
},
|
||||
"bedrock/*/1-month-commitment/cohere.command-light-text-v14": {
|
||||
"cost_per_second": 0.001902,
|
||||
"input_cost_per_second": 0.001902,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.001902,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/*/1-month-commitment/cohere.command-text-v14": {
|
||||
"cost_per_second": 0.011,
|
||||
"input_cost_per_second": 0.011,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.011,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/*/6-month-commitment/cohere.command-light-text-v14": {
|
||||
"cost_per_second": 0.0011416,
|
||||
"input_cost_per_second": 0.0011416,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0011416,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/*/6-month-commitment/cohere.command-text-v14": {
|
||||
"cost_per_second": 0.0066027,
|
||||
"input_cost_per_second": 0.0066027,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0066027,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/guardrails": {
|
||||
|
|
@ -12737,61 +12737,61 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.01475,
|
||||
"input_cost_per_second": 0.01475,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.01475,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.0455,
|
||||
"input_cost_per_second": 0.0455,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0455
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.0455,
|
||||
"input_cost_per_second": 0.0455,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0455,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.008194,
|
||||
"input_cost_per_second": 0.008194,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.008194,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.02527,
|
||||
"input_cost_per_second": 0.02527,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.02527
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.02527,
|
||||
"input_cost_per_second": 0.02527,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.02527,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/anthropic.claude-instant-v1": {
|
||||
|
|
@ -13241,61 +13241,61 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.01635,
|
||||
"input_cost_per_second": 0.01635,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.01635,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.0415,
|
||||
"input_cost_per_second": 0.0415,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0415
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.0415,
|
||||
"input_cost_per_second": 0.0415,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0415,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.009083,
|
||||
"input_cost_per_second": 0.009083,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.009083,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.02305,
|
||||
"input_cost_per_second": 0.02305,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.02305
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.02305,
|
||||
"input_cost_per_second": 0.02305,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.02305,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/anthropic.claude-instant-v1": {
|
||||
|
|
@ -13737,61 +13737,61 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.011,
|
||||
"input_cost_per_second": 0.011,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.011,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.0175,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0175
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.0175,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0175,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.00611,
|
||||
"input_cost_per_second": 0.00611,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00611,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.00972,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00972
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.00972,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00972,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/anthropic.claude-instant-v1": {
|
||||
|
|
@ -14385,61 +14385,61 @@
|
|||
"output_cost_per_token": 6e-07
|
||||
},
|
||||
"bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.011,
|
||||
"input_cost_per_second": 0.011,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.011,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.0175,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0175
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.0175,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.0175,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1": {
|
||||
"cost_per_second": 0.00611,
|
||||
"input_cost_per_second": 0.00611,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00611,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v1": {
|
||||
"cost_per_second": 0.00972,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00972
|
||||
"mode": "chat"
|
||||
},
|
||||
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1": {
|
||||
"cost_per_second": 0.00972,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"max_tokens": 8191,
|
||||
"mode": "chat",
|
||||
"output_cost_per_second": 0.00972,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/anthropic.claude-instant-v1": {
|
||||
|
|
@ -38527,7 +38527,6 @@
|
|||
},
|
||||
"mistral/voxtral-small-2507": {
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"input_cost_per_second": 6.666666666666667e-05,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 32768,
|
||||
|
|
@ -38543,7 +38542,6 @@
|
|||
},
|
||||
"mistral/voxtral-small-latest": {
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"input_cost_per_second": 6.666666666666667e-05,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "mistral",
|
||||
"max_input_tokens": 32768,
|
||||
|
|
|
|||
|
|
@ -249,6 +249,10 @@
|
|||
"comment": {
|
||||
"type": "string"
|
||||
},
|
||||
"cost_per_second": {
|
||||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"default_reasoning_effort": {
|
||||
"type": "string",
|
||||
"description": "Reasoning effort the provider applies when the request omits reasoning_effort. Gates whether a non-default temperature or the top_p/logprobs sampling params are accepted, which hold only when the effort resolves to 'none'.",
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@ model_list:
|
|||
- model_name: sagemaker-completion-model
|
||||
litellm_params:
|
||||
model: sagemaker/berri-benchmarking-Llama-2-70b-chat-hf-4
|
||||
input_cost_per_second: 0.000420
|
||||
cost_per_second: 0.000420
|
||||
- model_name: text-embedding-ada-002
|
||||
litellm_params:
|
||||
model: openai/text-embedding-3-small
|
||||
|
|
|
|||
196
tests/integration/pricing/test_per_second_pricing.py
Normal file
196
tests/integration/pricing/test_per_second_pricing.py
Normal file
|
|
@ -0,0 +1,196 @@
|
|||
import json
|
||||
import uuid
|
||||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
from pydantic import JsonValue
|
||||
|
||||
from tests.integration._support.client import JSON_OBJECT, Gateway, eventually, object_value, string_value
|
||||
from tests.integration._support.database import read_rows
|
||||
from tests.integration._support.upstream import delete_scenario, register_scenario
|
||||
from tests.integration.cost_calculation.cost_tracking_case import SseResponse
|
||||
|
||||
RATE: Final = 0.5
|
||||
FRAME_DELAY_MS: Final = 300
|
||||
CONTENT: Final = ("one", " two", " three", " four")
|
||||
PRICING_FIELDS: Final = frozenset({"cost_per_second", "input_cost_per_second", "output_cost_per_second"})
|
||||
PER_SECOND_CONFIGURATIONS: Final[tuple[tuple[str, Mapping[str, JsonValue]], ...]] = (
|
||||
("new_field", {"cost_per_second": RATE}),
|
||||
("legacy_input", {"input_cost_per_second": RATE}),
|
||||
("legacy_output", {"output_cost_per_second": RATE}),
|
||||
("legacy_both", {"input_cost_per_second": RATE, "output_cost_per_second": 0.25}),
|
||||
(
|
||||
"all_three",
|
||||
{"cost_per_second": RATE, "input_cost_per_second": 0.25, "output_cost_per_second": 0.125},
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _sse_chunk(delta: dict[str, JsonValue], finish_reason: str | None) -> str:
|
||||
payload: Final = {
|
||||
"id": "$REQUEST_ID",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "integration-per-second",
|
||||
"choices": [{"index": 0, "delta": delta, "finish_reason": finish_reason}],
|
||||
}
|
||||
return f"data: {json.dumps(payload)}"
|
||||
|
||||
|
||||
def _sse_frames() -> tuple[str, ...]:
|
||||
content_frames: Final = tuple(_sse_chunk({"content": content}, None) for content in CONTENT)
|
||||
usage_payload: Final = {
|
||||
"id": "$REQUEST_ID",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1,
|
||||
"model": "integration-per-second",
|
||||
"choices": [],
|
||||
"usage": {"prompt_tokens": 20, "completion_tokens": 20, "total_tokens": 40},
|
||||
}
|
||||
usage_frame: Final = f"data: {json.dumps(usage_payload)}"
|
||||
return (*content_frames, _sse_chunk({}, "stop"), usage_frame, "data: [DONE]")
|
||||
|
||||
|
||||
def _stream_content(event: dict[str, JsonValue]) -> str:
|
||||
choices: Final = event.get("choices")
|
||||
if not isinstance(choices, list) or not choices:
|
||||
return ""
|
||||
delta: Final = object_value(object_value(choices[0])["delta"])
|
||||
content: Final = delta.get("content")
|
||||
return content if isinstance(content, str) else ""
|
||||
|
||||
|
||||
def _clear_observations(upstream: httpx.Client) -> None:
|
||||
response: Final = upstream.get("/__observations")
|
||||
assert response.status_code == 200, response.text
|
||||
|
||||
|
||||
def _observed_request_body(upstream: httpx.Client) -> dict[str, JsonValue]:
|
||||
observations: Final = JSON_OBJECT.validate_json(upstream.get("/__observations").content)["requests"]
|
||||
assert isinstance(observations, list)
|
||||
assert len(observations) == 1
|
||||
return object_value(object_value(observations[0])["body"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("pricing_case", "pricing"),
|
||||
PER_SECOND_CONFIGURATIONS,
|
||||
ids=("new_field", "legacy_input", "legacy_output", "legacy_both", "all_three"),
|
||||
)
|
||||
def test_chat_per_second_pricing_is_charged_once_and_not_forwarded(
|
||||
gateway: Gateway, pricing_case: str, pricing: Mapping[str, JsonValue]
|
||||
) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
scenario_id: Final = f"per-second-{pricing_case}-{uuid.uuid4().hex}"
|
||||
key: Final = scenario.key()
|
||||
model: Final = scenario.model(
|
||||
model=f"openai/integration-per-second-{uuid.uuid4().hex}",
|
||||
api_key=scenario_id,
|
||||
api_base=f"{gateway.upstream_url.rstrip('/')}/v1",
|
||||
**pricing,
|
||||
)
|
||||
with httpx.Client(base_url=gateway.upstream_url, trust_env=False) as upstream:
|
||||
_clear_observations(upstream)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/chat/completions",
|
||||
{"model": model, "messages": [{"role": "user", "content": "price this request"}]},
|
||||
key=key,
|
||||
)
|
||||
body: Final = _observed_request_body(upstream)
|
||||
assert response.status_code == 200, f"{pricing_case}: {response.text}"
|
||||
response_cost: Final = float(response.headers.get("x-litellm-response-cost", "0"))
|
||||
duration_ms: Final = float(response.headers.get("x-litellm-response-duration-ms", "0"))
|
||||
assert response_cost > 0, f"{pricing_case}: cost={response_cost}, duration_ms={duration_ms}, body={body}"
|
||||
assert response_cost == pytest.approx(RATE * duration_ms / 1000, rel=1e-3), (
|
||||
f"{pricing_case}: cost={response_cost}, duration_ms={duration_ms}, body={body}"
|
||||
)
|
||||
assert not PRICING_FIELDS.intersection(body), body
|
||||
|
||||
request_id: Final = string_value(object_value(response.json())["id"])
|
||||
rows: Final = eventually(
|
||||
lambda: read_rows(
|
||||
'SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id = %s',
|
||||
(request_id,),
|
||||
),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=70,
|
||||
)
|
||||
assert float(str(rows[0]["spend"])) == pytest.approx(response_cost, rel=1e-3)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("pricing_case", "pricing"),
|
||||
PER_SECOND_CONFIGURATIONS,
|
||||
ids=("new_field", "legacy_input", "legacy_output", "legacy_both", "all_three"),
|
||||
)
|
||||
def test_streaming_chat_per_second_pricing_covers_the_full_stream(
|
||||
gateway: Gateway, pricing_case: str, pricing: Mapping[str, JsonValue]
|
||||
) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
scenario_id: Final = f"per-second-stream-{pricing_case}-{uuid.uuid4().hex}"
|
||||
frames: Final = _sse_frames()
|
||||
handle: Final = register_scenario(
|
||||
scenario_id,
|
||||
SseResponse(content_type="text/event-stream", frames=frames, frame_delay_ms=FRAME_DELAY_MS),
|
||||
)
|
||||
scenario.cleanups.callback(delete_scenario, handle)
|
||||
key: Final = scenario.key()
|
||||
model: Final = scenario.model(
|
||||
model=f"openai/integration-per-second-{uuid.uuid4().hex}",
|
||||
api_key=scenario_id,
|
||||
api_base=handle.api_base(),
|
||||
**pricing,
|
||||
)
|
||||
with httpx.Client(base_url=gateway.upstream_url, trust_env=False) as upstream:
|
||||
_clear_observations(upstream)
|
||||
with gateway.client.stream(
|
||||
"POST",
|
||||
"/v1/chat/completions",
|
||||
json={
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "price this streamed request"}],
|
||||
"stream": True,
|
||||
"stream_options": {"include_usage": True},
|
||||
},
|
||||
headers={"Authorization": f"Bearer {key}"},
|
||||
) as response:
|
||||
stream_lines: Final = tuple(response.iter_lines())
|
||||
assert response.status_code == 200, "\n".join(stream_lines)
|
||||
body: Final = _observed_request_body(upstream)
|
||||
|
||||
events: Final = tuple(
|
||||
JSON_OBJECT.validate_json(line.removeprefix("data: "))
|
||||
for line in stream_lines
|
||||
if line.startswith("data: ") and line != "data: [DONE]"
|
||||
)
|
||||
assert len(events) == len(frames) - 1, events
|
||||
assert "".join(_stream_content(event) for event in events) == "".join(CONTENT), events
|
||||
usage: Final = object_value(events[-1]["usage"])
|
||||
assert usage["total_tokens"] == 40, events[-1]
|
||||
request_id: Final = string_value(events[0]["id"])
|
||||
rows: Final = eventually(
|
||||
lambda: read_rows(
|
||||
'SELECT spend, request_duration_ms, '
|
||||
'CAST(EXTRACT(EPOCH FROM ("endTime" - "startTime")) * 1000 AS DOUBLE PRECISION) '
|
||||
'AS elapsed_duration_ms '
|
||||
'FROM "LiteLLM_SpendLogs" WHERE request_id = %s',
|
||||
(request_id,),
|
||||
),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=70,
|
||||
)
|
||||
spend: Final = float(str(rows[0]["spend"]))
|
||||
request_duration_ms: Final = float(str(rows[0]["request_duration_ms"]))
|
||||
elapsed_duration_ms: Final = float(str(rows[0]["elapsed_duration_ms"]))
|
||||
assert spend == pytest.approx(RATE * request_duration_ms / 1000, rel=5e-2), (
|
||||
f"spend={spend}, request_duration_ms={request_duration_ms}, "
|
||||
f"endTime-startTime duration_ms={elapsed_duration_ms}, body={body}"
|
||||
)
|
||||
total_frame_delay_seconds: Final = (len(frames) - 1) * FRAME_DELAY_MS / 1000
|
||||
assert spend >= RATE * total_frame_delay_seconds * 0.95, (
|
||||
f"spend={spend}, total frame delay={total_frame_delay_seconds}s, body={body}"
|
||||
)
|
||||
assert not PRICING_FIELDS.intersection(body), body
|
||||
|
|
@ -713,7 +713,7 @@ def test_sagemaker_embeddings():
|
|||
response = litellm.embedding(
|
||||
model="sagemaker/berri-benchmarking-gpt-j-6b-fp16",
|
||||
input=["good morning from litellm", "this is another item"],
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
)
|
||||
print(f"response: {response}")
|
||||
cost = completion_cost(completion_response=response)
|
||||
|
|
@ -731,7 +731,7 @@ async def test_sagemaker_aembeddings():
|
|||
response = await litellm.aembedding(
|
||||
model="sagemaker/berri-benchmarking-gpt-j-6b-fp16",
|
||||
input=["good morning from litellm", "this is another item"],
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
)
|
||||
print(f"response: {response}")
|
||||
cost = completion_cost(completion_response=response)
|
||||
|
|
|
|||
|
|
@ -55,7 +55,7 @@ async def test_completion_sagemaker(sync_mode):
|
|||
],
|
||||
temperature=0.2,
|
||||
max_tokens=80,
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
)
|
||||
else:
|
||||
response = await litellm.acompletion(
|
||||
|
|
@ -65,7 +65,7 @@ async def test_completion_sagemaker(sync_mode):
|
|||
],
|
||||
temperature=0.2,
|
||||
max_tokens=80,
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
print(response)
|
||||
|
|
@ -169,7 +169,7 @@ async def test_completion_sagemaker_stream(sync_mode, model):
|
|||
temperature=0.2,
|
||||
stream=True,
|
||||
max_tokens=80,
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
)
|
||||
|
||||
for idx, chunk in enumerate(response):
|
||||
|
|
@ -187,7 +187,7 @@ async def test_completion_sagemaker_stream(sync_mode, model):
|
|||
stream=True,
|
||||
temperature=0.2,
|
||||
max_tokens=80,
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
)
|
||||
|
||||
print("streaming response")
|
||||
|
|
@ -280,7 +280,7 @@ async def test_acompletion_sagemaker_non_stream():
|
|||
],
|
||||
temperature=0.2,
|
||||
max_tokens=80,
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
)
|
||||
|
||||
# Print what was called on the mock
|
||||
|
|
@ -340,7 +340,7 @@ async def test_completion_sagemaker_non_stream():
|
|||
],
|
||||
temperature=0.2,
|
||||
max_tokens=80,
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
)
|
||||
|
||||
# Print what was called on the mock
|
||||
|
|
@ -457,7 +457,7 @@ async def test_completion_sagemaker_non_stream_with_aws_params():
|
|||
],
|
||||
temperature=0.2,
|
||||
max_tokens=80,
|
||||
input_cost_per_second=0.000420,
|
||||
cost_per_second=0.000420,
|
||||
aws_access_key_id="gm",
|
||||
aws_secret_access_key="s",
|
||||
aws_region_name="us-west-5",
|
||||
|
|
|
|||
|
|
@ -8697,7 +8697,7 @@ def test_model_has_no_cost_mapping_non_token_price_from_litellm_params_is_false(
|
|||
assert model_has_no_cost_mapping(model="custom-tts", llm_router=router) is False
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cost_field", ["input_cost_per_second", "input_cost_per_token"])
|
||||
@pytest.mark.parametrize("cost_field", ["cost_per_second", "input_cost_per_second", "input_cost_per_token"])
|
||||
def test_model_has_no_cost_mapping_explicit_zero_price_is_false(cost_field):
|
||||
from litellm.proxy.auth.auth_checks import model_has_no_cost_mapping
|
||||
from litellm.router import Router
|
||||
|
|
|
|||
|
|
@ -65,6 +65,7 @@ class TestStripClientPricingOverrides:
|
|||
for field in (
|
||||
"input_cost_per_token",
|
||||
"output_cost_per_token",
|
||||
"cost_per_second",
|
||||
"input_cost_per_second",
|
||||
"cache_creation_input_token_cost",
|
||||
):
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ from litellm.litellm_core_utils.llm_cost_calc.zero_cost_diagnostic import (
|
|||
)
|
||||
from litellm.types.utils import CompletionTokensDetailsWrapper, PromptTokensDetailsWrapper, Usage
|
||||
|
||||
PER_SECOND_ENTRY: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042}
|
||||
PER_SECOND_ENTRY: Final = {"cost_per_second": 0.00042}
|
||||
FREE_ENTRY: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0, "cache_read_input_token_cost": 2e-08}
|
||||
PRICED_ENTRY: Final = {"input_cost_per_token": 1e-06, "output_cost_per_token": 2e-06}
|
||||
TEXT_USAGE: Final = Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
|
|
|
|||
|
|
@ -607,8 +607,7 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_
|
|||
litellm.register_model(
|
||||
model_cost={
|
||||
deployment_id: {
|
||||
"input_cost_per_second": 0.02,
|
||||
"output_cost_per_second": 0.04,
|
||||
"cost_per_second": 0.02,
|
||||
"litellm_provider": "openai",
|
||||
"mode": "chat",
|
||||
}
|
||||
|
|
@ -627,8 +626,7 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_
|
|||
logging_obj.update_environment_variables(
|
||||
model="gpt-5.4-nano",
|
||||
litellm_params={
|
||||
"input_cost_per_second": 0.02,
|
||||
"output_cost_per_second": 0.04,
|
||||
"cost_per_second": 0.02,
|
||||
"metadata": {"model_info": {"id": deployment_id}},
|
||||
},
|
||||
optional_params={},
|
||||
|
|
@ -650,4 +648,4 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_
|
|||
)
|
||||
|
||||
assert result._response_ms == pytest.approx(2000)
|
||||
assert result._hidden_params["response_cost"] == pytest.approx((0.02 + 0.04) * 2)
|
||||
assert result._hidden_params["response_cost"] == pytest.approx(0.02 * 2)
|
||||
|
|
|
|||
|
|
@ -21,7 +21,13 @@ from litellm.litellm_core_utils.get_litellm_params import (
|
|||
from litellm.types.litellm_params import ControlOptions
|
||||
|
||||
NAMED_PRICE_PARAMS: Final = frozenset(
|
||||
{"input_cost_per_token", "output_cost_per_token", "input_cost_per_second", "output_cost_per_second"}
|
||||
{
|
||||
"input_cost_per_token",
|
||||
"output_cost_per_token",
|
||||
"cost_per_second",
|
||||
"input_cost_per_second",
|
||||
"output_cost_per_second",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -495,7 +495,7 @@ class TestZeroCostDiagnostic:
|
|||
DEPLOYMENT_ID: Final = "lit7898-query-only-priced-deployment"
|
||||
MODEL_GROUP: Final = "query-only-priced-chat"
|
||||
QUERY_ONLY_PRICING: Final = {"input_cost_per_query": 0.00042}
|
||||
PER_SECOND_PRICING: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042}
|
||||
PER_SECOND_PRICING: Final = {"cost_per_second": 0.00042}
|
||||
FREE_PRICING: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0}
|
||||
|
||||
@pytest.fixture(params=["query_only", "free"])
|
||||
|
|
@ -845,7 +845,7 @@ class TestZeroCostDiagnostic:
|
|||
response: Final = self._response(usage)
|
||||
response._response_ms = 1000.0
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
||||
assert logging_obj._response_cost_calculator(result=response) == pytest.approx(0.00084)
|
||||
assert logging_obj._response_cost_calculator(result=response) == pytest.approx(0.00042)
|
||||
|
||||
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
|
||||
assert self._zero_cost_warnings(caplog) == []
|
||||
|
|
|
|||
|
|
@ -5043,6 +5043,7 @@ class TestRouterPreRoutingAliasOverrides:
|
|||
"model": "auto_router/complexity_router",
|
||||
"input_cost_per_token": 0.0,
|
||||
"output_cost_per_token": 0.0,
|
||||
"cost_per_second": 0.0,
|
||||
"input_cost_per_second": 0.0,
|
||||
"drop_params": True,
|
||||
"complexity_router_config": {"tiers": {"SIMPLE": "gpt-4o-mini"}},
|
||||
|
|
@ -5064,7 +5065,12 @@ class TestRouterPreRoutingAliasOverrides:
|
|||
assert result is not None
|
||||
# Non-pricing alias params still carry over.
|
||||
assert request_kwargs["drop_params"] is True
|
||||
for field in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second"):
|
||||
for field in (
|
||||
"input_cost_per_token",
|
||||
"output_cost_per_token",
|
||||
"cost_per_second",
|
||||
"input_cost_per_second",
|
||||
):
|
||||
assert field not in request_kwargs
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
|
|||
|
|
@ -3037,9 +3037,9 @@ def test_completion_cost_logs_cache_and_reasoning_breakdown_for_custom_pricing()
|
|||
@pytest.mark.parametrize("custom_llm_provider", ["together_ai", "openai", "anthropic", "bedrock", "azure"])
|
||||
def test_cost_per_token_per_second_pricing(monkeypatch, custom_llm_provider: str):
|
||||
"""
|
||||
Models priced by duration (input/output_cost_per_second) with no per-token rates
|
||||
Models priced by input/output duration rates with no per-token rates
|
||||
must be billed as cost_per_second * response_time_ms / 1000 in cost_per_token,
|
||||
whether or not the provider has its own cost calculator.
|
||||
using only the input rate even when both are set, whether or not the provider has its own calculator.
|
||||
"""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
|
@ -3064,11 +3064,40 @@ def test_cost_per_token_per_second_pricing(monkeypatch, custom_llm_provider: str
|
|||
response_time_ms=1500.0,
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(0.02 * 1.5)
|
||||
assert completion_cost_value == pytest.approx(0.04 * 1.5)
|
||||
assert (prompt_cost, completion_cost_value) == pytest.approx((0.02 * 1.5, 0.0))
|
||||
|
||||
|
||||
def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(monkeypatch):
|
||||
def test_azure_chat_uses_token_rates_when_output_cost_per_second_is_set(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
model: Final = "test-azure-chat-token-and-output-second-pricing"
|
||||
litellm.register_model(
|
||||
model_cost={
|
||||
model: {
|
||||
"input_cost_per_token": 1e-6,
|
||||
"output_cost_per_token": 2e-6,
|
||||
"output_cost_per_second": 0.4,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "chat",
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
cost: Final = cost_per_token(
|
||||
model=model,
|
||||
custom_llm_provider="azure",
|
||||
prompt_tokens=10,
|
||||
completion_tokens=20,
|
||||
response_time_ms=1500.0,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx((10 * 1e-6, 20 * 2e-6))
|
||||
|
||||
|
||||
def test_cost_per_token_ignores_cost_per_second_when_token_pricing_is_set(monkeypatch):
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
|
|
@ -3078,8 +3107,7 @@ def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(m
|
|||
model: {
|
||||
"input_cost_per_token": 1e-6,
|
||||
"output_cost_per_token": 2e-6,
|
||||
"input_cost_per_second": 0.02,
|
||||
"output_cost_per_second": 0.04,
|
||||
"cost_per_second": 0.02,
|
||||
"litellm_provider": "openai",
|
||||
"mode": "chat",
|
||||
}
|
||||
|
|
@ -3098,6 +3126,39 @@ def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(m
|
|||
assert completion_cost_value == pytest.approx(20 * 2e-6)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("pricing_fields", "expected_rate"),
|
||||
[
|
||||
({"cost_per_second": 0.02}, 0.02),
|
||||
({"output_cost_per_second": 0.04}, 0.04),
|
||||
(
|
||||
{"cost_per_second": 0.05, "input_cost_per_second": 0.02, "output_cost_per_second": 0.04},
|
||||
0.05,
|
||||
),
|
||||
({"input_cost_per_second": 0.02}, 0.02),
|
||||
],
|
||||
)
|
||||
def test_cost_per_token_resolves_per_second_rate_precedence(
|
||||
monkeypatch, pricing_fields: dict[str, float], expected_rate: float
|
||||
):
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
model: Final = "test-chat-per-second-rate-precedence"
|
||||
entry: Final = {**pricing_fields, "litellm_provider": "together_ai", "mode": "chat"}
|
||||
litellm.register_model(
|
||||
model_cost={model: entry}
|
||||
)
|
||||
|
||||
assert cost_per_token(
|
||||
model=model,
|
||||
custom_llm_provider="together_ai",
|
||||
prompt_tokens=10,
|
||||
completion_tokens=20,
|
||||
response_time_ms=1500.0,
|
||||
) == pytest.approx((expected_rate * 1.5, 0.0))
|
||||
|
||||
|
||||
def _logging_obj_with_call_window(duration_ms: float) -> Logging:
|
||||
start_time: Final = datetime.datetime(2026, 9, 21, 12, 0, 0)
|
||||
logging_obj: Final = Logging(
|
||||
|
|
@ -3160,7 +3221,7 @@ def test_completion_cost_per_second_deployment_bills_the_call_duration(
|
|||
litellm_logging_obj=_logging_obj_with_call_window(logged_duration_ms),
|
||||
)
|
||||
|
||||
assert cost == pytest.approx((0.02 + 0.04) * expected_seconds)
|
||||
assert cost == pytest.approx(0.02 * expected_seconds)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("mode", ["audio_transcription", "audio_speech", "video_generation", "realtime"])
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ calculations for DB-sourced models with prompt caching pricing.
|
|||
|
||||
import copy
|
||||
import os
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -993,3 +994,21 @@ def test_completion_cost_applies_off_peak_only_deployment_pricing():
|
|||
finally:
|
||||
_restore_model_cost_entries(original_entries)
|
||||
del router
|
||||
|
||||
|
||||
def test_completion_registers_cost_per_second_pricing():
|
||||
model_key: Final = "openai/test-cost-per-second-registration"
|
||||
original_entries: Final = _snapshot_model_cost_entries([model_key])
|
||||
|
||||
try:
|
||||
litellm.completion(
|
||||
model=model_key,
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
api_key="fake-key",
|
||||
cost_per_second=0.02,
|
||||
mock_response="hello back",
|
||||
)
|
||||
|
||||
assert litellm.model_cost[model_key]["cost_per_second"] == 0.02
|
||||
finally:
|
||||
_restore_model_cost_entries(original_entries)
|
||||
|
|
|
|||
|
|
@ -648,6 +648,7 @@ def validate_model_cost_values(model_data, exceptions=None):
|
|||
"output_cost_per_image_4K",
|
||||
"input_cost_per_pixel",
|
||||
"output_cost_per_pixel",
|
||||
"cost_per_second",
|
||||
"input_cost_per_second",
|
||||
"output_cost_per_second",
|
||||
"output_cost_per_second_480p",
|
||||
|
|
@ -829,6 +830,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"input_cost_per_pixel": {"type": "number"},
|
||||
"input_cost_per_query": {"type": "number"},
|
||||
"input_cost_per_request": {"type": "number"},
|
||||
"cost_per_second": {"type": "number"},
|
||||
"input_cost_per_second": {"type": "number"},
|
||||
"input_cost_per_token": {"type": "number"},
|
||||
"input_cost_per_token_above_128k_tokens": {"type": "number"},
|
||||
|
|
|
|||
|
|
@ -40,6 +40,7 @@ def test_custom_pricing_params_keeps_every_field_it_had():
|
|||
"output_cost_per_character",
|
||||
"cache_read_input_token_cost",
|
||||
"cache_creation_input_token_cost",
|
||||
"cost_per_second",
|
||||
"input_cost_per_second",
|
||||
"cache_read_input_token_cost_flex",
|
||||
"input_cost_per_character_above_128k_tokens",
|
||||
|
|
|
|||
4
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
4
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -33013,6 +33013,8 @@ export interface components {
|
|||
complexity_router_default_model?: string | null;
|
||||
/** Configurable Clientside Auth Params */
|
||||
configurable_clientside_auth_params?: (string | components["schemas"]["ConfigurableClientsideParamsCustomAuth-Input"])[] | null;
|
||||
/** Cost Per Second */
|
||||
cost_per_second?: number | null;
|
||||
/** Custom Llm Provider */
|
||||
custom_llm_provider?: string | null;
|
||||
/** Default Api Key Rpm Limit */
|
||||
|
|
@ -46842,6 +46844,8 @@ export interface components {
|
|||
complexity_router_default_model?: string | null;
|
||||
/** Configurable Clientside Auth Params */
|
||||
configurable_clientside_auth_params?: (string | components["schemas"]["ConfigurableClientsideParamsCustomAuth-Input"])[] | null;
|
||||
/** Cost Per Second */
|
||||
cost_per_second?: number | null;
|
||||
/** Custom Llm Provider */
|
||||
custom_llm_provider?: string | null;
|
||||
/** Default Api Key Rpm Limit */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue