fix(cost_calculator): bill chat per-second pricing once with a new cost_per_second field (#43614)

* feat(cost_calculator): add cost_per_second for chat per-second pricing

Keep legacy input_cost_per_second and output_cost_per_second as aliases for chat, completion, embedding and responses. When both legacy fields are set, input_cost_per_second wins

Move Bedrock commitment rows to cost_per_second so they bill once

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(cost_calculator): drop legacy per-second fields from chat paths

Keep Azure chat token pricing generic and update inert Voxtral rates and SageMaker examples

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(cost_calculator): recognize output-only per-second rates

Include output_cost_per_second when checking whether a deployment cost entry has pricing so output-only legacy aliases remain attached to the deployment during cost selection

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(pricing): cover cost_per_second and legacy per-second aliases through the proxy

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(cost_calculator): drop output_cost_per_second as a chat per-second alias

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* feat(cost_calculator): restore output_cost_per_second as a chat per-second fallback

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(cost-map): keep input_cost_per_second on bedrock commitment rows for older clients

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-09-29 11:27:14 -07:00 • committed by GitHub
parent 0fe4028cd9
commit abc85c2651
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
29 changed files with 441 additions and 140 deletions

View file

@ -24,7 +24,7 @@ model_list:
- model_name: sagemaker-completion-model
litellm_params:
model: sagemaker/berri-benchmarking-Llama-2-70b-chat-hf-4
input_cost_per_second: 0.000420
cost_per_second: 0.000420
- model_name: text-embedding-ada-002
litellm_params:
model: azure/azure-embedding-model

View file

@ -125,6 +125,8 @@ pub struct ModelInfo {
pub computer_use_input_cost_per_1k_tokens: Option<f64>,
#[serde(skip_serializing_if = "Option::is_none")]
pub computer_use_output_cost_per_1k_tokens: Option<f64>,
#[serde(skip_serializing_if = "Option::is_none")]
pub cost_per_second: Option<f64>,
/// Reasoning effort the provider applies when the request omits reasoning_effort. Gates whether a non-default temperature or the top_p/logprobs sampling params are accepted, which hold only when the effort resolves to 'none'.
#[serde(skip_serializing_if = "Option::is_none")]
pub default_reasoning_effort: Option<ReasoningEffort>,

View file

@ -351,19 +351,27 @@ def _per_second_pricing_cost(
return None
if _has_token_or_tiered_pricing(model_info) or not _bills_wall_clock_seconds(model_info):
return None
cost_per_second: Final = model_info.get("cost_per_second")
input_cost_per_second: Final = model_info.get("input_cost_per_second")
output_cost_per_second: Final = model_info.get("output_cost_per_second")
if input_cost_per_second is None and output_cost_per_second is None:
resolved_cost_per_second: Final = (
cost_per_second
if cost_per_second is not None
else input_cost_per_second
if input_cost_per_second is not None
else output_cost_per_second
)
if resolved_cost_per_second is None:
return None
seconds: Final = (response_time_ms or 0.0) / 1000
verbose_logger.debug(
"For model=%s - input_cost_per_second: %s; output_cost_per_second: %s; response time: %s",
"For model=%s - cost_per_second: %s; response time: %s",
model,
input_cost_per_second,
output_cost_per_second,
resolved_cost_per_second,
response_time_ms,
)
return (input_cost_per_second or 0.0) * seconds, (output_cost_per_second or 0.0) * seconds
return resolved_cost_per_second * seconds, 0.0
def cost_per_token(
@ -790,7 +798,9 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None
return value if isinstance(value, str) and value else None
_NON_TOKEN_RATE_FIELDS: Final = frozenset({"input_cost_per_second", "input_cost_per_query", "tiered_pricing"})
_NON_TOKEN_RATE_FIELDS: Final = frozenset(
{"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"}
)
def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool:

View file

@ -155,6 +155,7 @@ def get_litellm_params(
allm_passthrough_route=None,
preset_cache_key=None,
no_log=None,
cost_per_second: float | None = None,
input_cost_per_second=None,
input_cost_per_token=None,
output_cost_per_token=None,
@ -216,6 +217,7 @@ def get_litellm_params(
"preset_cache_key": preset_cache_key,
"no-log": no_log or kwargs.get("no-log"),
"stream_response": {}, # litellm_call_id: ModelResponse Dict
"cost_per_second": cost_per_second,
"input_cost_per_token": input_cost_per_token,
"input_cost_per_second": input_cost_per_second,
"output_cost_per_token": output_cost_per_token,

View file

@ -3,12 +3,8 @@ Helper util for handling azure openai-specific cost calculation
- e.g.: prompt caching, audio tokens
"""
from typing import Final
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.types.utils import Usage
from litellm.utils import get_model_info
def cost_per_token(
@ -27,26 +23,6 @@ def cost_per_token(
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
"""
## GET MODEL INFO
model_info: Final = get_model_info(model=model, custom_llm_provider="azure")
## Speech / Audio cost calculation (cost per second for TTS models)
if (
"output_cost_per_second" in model_info
and model_info["output_cost_per_second"] is not None
and response_time_ms is not None
):
verbose_logger.debug(
"For model=%s - output_cost_per_second: %s; response time: %s",
model,
model_info.get("output_cost_per_second"),
response_time_ms,
)
## COST PER SECOND ##
prompt_cost: Final = 0.0
completion_cost: Final = model_info["output_cost_per_second"] * response_time_ms / 1000
return prompt_cost, completion_cost
## Use generic cost calculator for all other cases
## This properly handles: text tokens, audio tokens, cached tokens, reasoning tokens, etc.
return generic_cost_per_token(

View file

@ -5353,6 +5353,7 @@ def completion(
### CUSTOM MODEL COST ###
input_cost_per_token: Final = kwargs.get("input_cost_per_token", None)
output_cost_per_token: Final = kwargs.get("output_cost_per_token", None)
cost_per_second: Final = kwargs.get("cost_per_second", None)
input_cost_per_second: Final = kwargs.get("input_cost_per_second", None)
output_cost_per_second: Final = kwargs.get("output_cost_per_second", None)
### CUSTOM PROMPT TEMPLATE ###
@ -5514,8 +5515,11 @@ def completion(
### REGISTER CUSTOM MODEL PRICING -- IF GIVEN ###
if (
input_cost_per_token is not None and output_cost_per_token is not None
) or input_cost_per_second is not None:
(input_cost_per_token is not None and output_cost_per_token is not None)
or input_cost_per_second is not None
or output_cost_per_second is not None
or cost_per_second is not None
):
_register_custom_pricing_for_request(
model=model,
custom_llm_provider=custom_llm_provider,
@ -5657,6 +5661,7 @@ def completion(
proxy_server_request=proxy_server_request,
preset_cache_key=preset_cache_key,
no_log=no_log,
cost_per_second=cost_per_second,
input_cost_per_second=input_cost_per_second,
input_cost_per_token=input_cost_per_token,
output_cost_per_second=output_cost_per_second,
@ -6354,7 +6359,9 @@ def embedding(
### CUSTOM MODEL COST ###
input_cost_per_token: Final = kwargs.get("input_cost_per_token", None)
output_cost_per_token: Final = kwargs.get("output_cost_per_token", None)
cost_per_second: Final = kwargs.get("cost_per_second", None)
input_cost_per_second: Final = kwargs.get("input_cost_per_second", None)
output_cost_per_second: Final = kwargs.get("output_cost_per_second", None)
openai_params: Final = [
"user",
"dimensions",
@ -6395,7 +6402,12 @@ def embedding(
)
### REGISTER CUSTOM MODEL PRICING -- IF GIVEN ###
if (input_cost_per_token is not None and output_cost_per_token is not None) or input_cost_per_second is not None:
if (
(input_cost_per_token is not None and output_cost_per_token is not None)
or input_cost_per_second is not None
or output_cost_per_second is not None
or cost_per_second is not None
):
_register_custom_pricing_for_request(
model=model,
custom_llm_provider=custom_llm_provider,

View file

@ -12682,43 +12682,43 @@
"source": "https://developers.openai.com/api/docs/pricing"
},
"bedrock/*/1-month-commitment/cohere.command-light-text-v14": {
"cost_per_second": 0.001902,
"input_cost_per_second": 0.001902,
"litellm_provider": "bedrock",
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_second": 0.001902,
"supports_tool_choice": true
},
"bedrock/*/1-month-commitment/cohere.command-text-v14": {
"cost_per_second": 0.011,
"input_cost_per_second": 0.011,
"litellm_provider": "bedrock",
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_second": 0.011,
"supports_tool_choice": true
},
"bedrock/*/6-month-commitment/cohere.command-light-text-v14": {
"cost_per_second": 0.0011416,
"input_cost_per_second": 0.0011416,
"litellm_provider": "bedrock",
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_second": 0.0011416,
"supports_tool_choice": true
},
"bedrock/*/6-month-commitment/cohere.command-text-v14": {
"cost_per_second": 0.0066027,
"input_cost_per_second": 0.0066027,
"litellm_provider": "bedrock",
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_second": 0.0066027,
"supports_tool_choice": true
},
"bedrock/guardrails": {
@ -12737,61 +12737,61 @@
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.01475,
"input_cost_per_second": 0.01475,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.01475,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.0455,
"input_cost_per_second": 0.0455,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0455
"mode": "chat"
},
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.0455,
"input_cost_per_second": 0.0455,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0455,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.008194,
"input_cost_per_second": 0.008194,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.008194,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.02527,
"input_cost_per_second": 0.02527,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.02527
"mode": "chat"
},
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.02527,
"input_cost_per_second": 0.02527,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.02527,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/anthropic.claude-instant-v1": {
@ -13241,61 +13241,61 @@
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.01635,
"input_cost_per_second": 0.01635,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.01635,
"supports_tool_choice": true
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.0415,
"input_cost_per_second": 0.0415,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0415
"mode": "chat"
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.0415,
"input_cost_per_second": 0.0415,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0415,
"supports_tool_choice": true
},
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.009083,
"input_cost_per_second": 0.009083,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.009083,
"supports_tool_choice": true
},
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.02305,
"input_cost_per_second": 0.02305,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.02305
"mode": "chat"
},
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.02305,
"input_cost_per_second": 0.02305,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.02305,
"supports_tool_choice": true
},
"bedrock/eu-central-1/anthropic.claude-instant-v1": {
@ -13737,61 +13737,61 @@
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.011,
"input_cost_per_second": 0.011,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.011,
"supports_tool_choice": true
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.0175,
"input_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0175
"mode": "chat"
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.0175,
"input_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0175,
"supports_tool_choice": true
},
"bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.00611,
"input_cost_per_second": 0.00611,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00611,
"supports_tool_choice": true
},
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.00972,
"input_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00972
"mode": "chat"
},
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.00972,
"input_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00972,
"supports_tool_choice": true
},
"bedrock/us-east-1/anthropic.claude-instant-v1": {
@ -14385,61 +14385,61 @@
"output_cost_per_token": 6e-07
},
"bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.011,
"input_cost_per_second": 0.011,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.011,
"supports_tool_choice": true
},
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.0175,
"input_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0175
"mode": "chat"
},
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.0175,
"input_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0175,
"supports_tool_choice": true
},
"bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.00611,
"input_cost_per_second": 0.00611,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00611,
"supports_tool_choice": true
},
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.00972,
"input_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00972
"mode": "chat"
},
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.00972,
"input_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00972,
"supports_tool_choice": true
},
"bedrock/us-west-2/anthropic.claude-instant-v1": {
@ -38527,7 +38527,6 @@
},
"mistral/voxtral-small-2507": {
"cache_read_input_token_cost": 1e-08,
"input_cost_per_second": 6.666666666666667e-05,
"input_cost_per_token": 1e-07,
"litellm_provider": "mistral",
"max_input_tokens": 32768,
@ -38543,7 +38542,6 @@
},
"mistral/voxtral-small-latest": {
"cache_read_input_token_cost": 1e-08,
"input_cost_per_second": 6.666666666666667e-05,
"input_cost_per_token": 1e-07,
"litellm_provider": "mistral",
"max_input_tokens": 32768,

View file

@ -8801,7 +8801,7 @@ class Router:
return
if any(
model_info.get(field) is not None
for field in ("input_cost_per_token", "input_cost_per_second", "tiered_pricing")
for field in ("input_cost_per_token", "input_cost_per_second", "cost_per_second", "tiered_pricing")
):
return
try:

View file

@ -606,6 +606,7 @@ class LiteLLMParamsTypedDict(TypedDict, total=False):
## CUSTOM PRICING ##
input_cost_per_token: float | None
output_cost_per_token: float | None
cost_per_second: ReadOnly[float | None]
input_cost_per_second: float | None
output_cost_per_second: float | None
output_cost_per_second_480p: ReadOnly[float | None]

View file

@ -329,6 +329,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
input_cost_per_video_per_second: float | None # only for vertex ai models
input_cost_per_audio_token_batches: ReadOnly[float | None]
input_cost_per_image_token_batches: ReadOnly[float | None]
cost_per_second: ReadOnly[float | None]
input_cost_per_second: float | None # for OpenAI Speech models
input_cost_per_token_batches: float | None
input_cost_per_video_token_batches: ReadOnly[float | None]
@ -2784,6 +2785,7 @@ class LoggedLiteLLMParams(TypedDict, total=False):
acompletion: bool | None
preset_cache_key: str | None
no_log: bool | None
cost_per_second: ReadOnly[float | None]
input_cost_per_second: float | None
input_cost_per_token: float | None
output_cost_per_token: float | None
@ -3709,6 +3711,7 @@ class MirroredPricingParams(BaseModel):
class CustomPricingLiteLLMParams(MirroredPricingParams):
## CUSTOM PRICING ##
cost_per_second: float | None = None
input_cost_per_second: float | None = None
output_cost_per_second: float | None = None
output_cost_per_second_1080p: float | None = None

View file

@ -6168,6 +6168,7 @@ def _get_model_info_helper(
),
input_cost_per_token_above_512k_tokens=_model_info.get("input_cost_per_token_above_512k_tokens", None),
input_cost_per_query=_model_info.get("input_cost_per_query", None),
cost_per_second=_model_info.get("cost_per_second", None),
input_cost_per_second=_model_info.get("input_cost_per_second", None),
input_cost_per_audio_token=_model_info.get("input_cost_per_audio_token", None),
input_cost_per_image_token=_model_info.get("input_cost_per_image_token", None),

View file

@ -12682,43 +12682,43 @@
"source": "https://developers.openai.com/api/docs/pricing"
},
"bedrock/*/1-month-commitment/cohere.command-light-text-v14": {
"cost_per_second": 0.001902,
"input_cost_per_second": 0.001902,
"litellm_provider": "bedrock",
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_second": 0.001902,
"supports_tool_choice": true
},
"bedrock/*/1-month-commitment/cohere.command-text-v14": {
"cost_per_second": 0.011,
"input_cost_per_second": 0.011,
"litellm_provider": "bedrock",
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_second": 0.011,
"supports_tool_choice": true
},
"bedrock/*/6-month-commitment/cohere.command-light-text-v14": {
"cost_per_second": 0.0011416,
"input_cost_per_second": 0.0011416,
"litellm_provider": "bedrock",
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_second": 0.0011416,
"supports_tool_choice": true
},
"bedrock/*/6-month-commitment/cohere.command-text-v14": {
"cost_per_second": 0.0066027,
"input_cost_per_second": 0.0066027,
"litellm_provider": "bedrock",
"max_input_tokens": 4096,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_second": 0.0066027,
"supports_tool_choice": true
},
"bedrock/guardrails": {
@ -12737,61 +12737,61 @@
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.01475,
"input_cost_per_second": 0.01475,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.01475,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.0455,
"input_cost_per_second": 0.0455,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0455
"mode": "chat"
},
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.0455,
"input_cost_per_second": 0.0455,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0455,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.008194,
"input_cost_per_second": 0.008194,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.008194,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.02527,
"input_cost_per_second": 0.02527,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.02527
"mode": "chat"
},
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.02527,
"input_cost_per_second": 0.02527,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.02527,
"supports_tool_choice": true
},
"bedrock/ap-northeast-1/anthropic.claude-instant-v1": {
@ -13241,61 +13241,61 @@
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.01635,
"input_cost_per_second": 0.01635,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.01635,
"supports_tool_choice": true
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.0415,
"input_cost_per_second": 0.0415,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0415
"mode": "chat"
},
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.0415,
"input_cost_per_second": 0.0415,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0415,
"supports_tool_choice": true
},
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.009083,
"input_cost_per_second": 0.009083,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.009083,
"supports_tool_choice": true
},
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.02305,
"input_cost_per_second": 0.02305,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.02305
"mode": "chat"
},
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.02305,
"input_cost_per_second": 0.02305,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.02305,
"supports_tool_choice": true
},
"bedrock/eu-central-1/anthropic.claude-instant-v1": {
@ -13737,61 +13737,61 @@
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.011,
"input_cost_per_second": 0.011,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.011,
"supports_tool_choice": true
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.0175,
"input_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0175
"mode": "chat"
},
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.0175,
"input_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0175,
"supports_tool_choice": true
},
"bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.00611,
"input_cost_per_second": 0.00611,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00611,
"supports_tool_choice": true
},
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.00972,
"input_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00972
"mode": "chat"
},
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.00972,
"input_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00972,
"supports_tool_choice": true
},
"bedrock/us-east-1/anthropic.claude-instant-v1": {
@ -14385,61 +14385,61 @@
"output_cost_per_token": 6e-07
},
"bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.011,
"input_cost_per_second": 0.011,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.011,
"supports_tool_choice": true
},
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.0175,
"input_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0175
"mode": "chat"
},
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.0175,
"input_cost_per_second": 0.0175,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.0175,
"supports_tool_choice": true
},
"bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1": {
"cost_per_second": 0.00611,
"input_cost_per_second": 0.00611,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00611,
"supports_tool_choice": true
},
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v1": {
"cost_per_second": 0.00972,
"input_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00972
"mode": "chat"
},
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1": {
"cost_per_second": 0.00972,
"input_cost_per_second": 0.00972,
"litellm_provider": "bedrock",
"max_input_tokens": 100000,
"max_output_tokens": 8191,
"max_tokens": 8191,
"mode": "chat",
"output_cost_per_second": 0.00972,
"supports_tool_choice": true
},
"bedrock/us-west-2/anthropic.claude-instant-v1": {
@ -38527,7 +38527,6 @@
},
"mistral/voxtral-small-2507": {
"cache_read_input_token_cost": 1e-08,
"input_cost_per_second": 6.666666666666667e-05,
"input_cost_per_token": 1e-07,
"litellm_provider": "mistral",
"max_input_tokens": 32768,
@ -38543,7 +38542,6 @@
},
"mistral/voxtral-small-latest": {
"cache_read_input_token_cost": 1e-08,
"input_cost_per_second": 6.666666666666667e-05,
"input_cost_per_token": 1e-07,
"litellm_provider": "mistral",
"max_input_tokens": 32768,

View file

@ -249,6 +249,10 @@
"comment": {
"type": "string"
},
"cost_per_second": {
"type": "number",
"minimum": 0
},
"default_reasoning_effort": {
"type": "string",
"description": "Reasoning effort the provider applies when the request omits reasoning_effort. Gates whether a non-default temperature or the top_p/logprobs sampling params are accepted, which hold only when the effort resolves to 'none'.",

View file

@ -31,7 +31,7 @@ model_list:
- model_name: sagemaker-completion-model
litellm_params:
model: sagemaker/berri-benchmarking-Llama-2-70b-chat-hf-4
input_cost_per_second: 0.000420
cost_per_second: 0.000420
- model_name: text-embedding-ada-002
litellm_params:
model: openai/text-embedding-3-small

View file

@ -0,0 +1,196 @@
import json
import uuid
from collections.abc import Mapping
from typing import Final
import httpx
import pytest
from pydantic import JsonValue
from tests.integration._support.client import JSON_OBJECT, Gateway, eventually, object_value, string_value
from tests.integration._support.database import read_rows
from tests.integration._support.upstream import delete_scenario, register_scenario
from tests.integration.cost_calculation.cost_tracking_case import SseResponse
RATE: Final = 0.5
FRAME_DELAY_MS: Final = 300
CONTENT: Final = ("one", " two", " three", " four")
PRICING_FIELDS: Final = frozenset({"cost_per_second", "input_cost_per_second", "output_cost_per_second"})
PER_SECOND_CONFIGURATIONS: Final[tuple[tuple[str, Mapping[str, JsonValue]], ...]] = (
("new_field", {"cost_per_second": RATE}),
("legacy_input", {"input_cost_per_second": RATE}),
("legacy_output", {"output_cost_per_second": RATE}),
("legacy_both", {"input_cost_per_second": RATE, "output_cost_per_second": 0.25}),
(
"all_three",
{"cost_per_second": RATE, "input_cost_per_second": 0.25, "output_cost_per_second": 0.125},
),
)
def _sse_chunk(delta: dict[str, JsonValue], finish_reason: str | None) -> str:
payload: Final = {
"id": "$REQUEST_ID",
"object": "chat.completion.chunk",
"created": 1,
"model": "integration-per-second",
"choices": [{"index": 0, "delta": delta, "finish_reason": finish_reason}],
}
return f"data: {json.dumps(payload)}"
def _sse_frames() -> tuple[str, ...]:
content_frames: Final = tuple(_sse_chunk({"content": content}, None) for content in CONTENT)
usage_payload: Final = {
"id": "$REQUEST_ID",
"object": "chat.completion.chunk",
"created": 1,
"model": "integration-per-second",
"choices": [],
"usage": {"prompt_tokens": 20, "completion_tokens": 20, "total_tokens": 40},
}
usage_frame: Final = f"data: {json.dumps(usage_payload)}"
return (*content_frames, _sse_chunk({}, "stop"), usage_frame, "data: [DONE]")
def _stream_content(event: dict[str, JsonValue]) -> str:
choices: Final = event.get("choices")
if not isinstance(choices, list) or not choices:
return ""
delta: Final = object_value(object_value(choices[0])["delta"])
content: Final = delta.get("content")
return content if isinstance(content, str) else ""
def _clear_observations(upstream: httpx.Client) -> None:
response: Final = upstream.get("/__observations")
assert response.status_code == 200, response.text
def _observed_request_body(upstream: httpx.Client) -> dict[str, JsonValue]:
observations: Final = JSON_OBJECT.validate_json(upstream.get("/__observations").content)["requests"]
assert isinstance(observations, list)
assert len(observations) == 1
return object_value(object_value(observations[0])["body"])
@pytest.mark.parametrize(
("pricing_case", "pricing"),
PER_SECOND_CONFIGURATIONS,
ids=("new_field", "legacy_input", "legacy_output", "legacy_both", "all_three"),
)
def test_chat_per_second_pricing_is_charged_once_and_not_forwarded(
gateway: Gateway, pricing_case: str, pricing: Mapping[str, JsonValue]
) -> None:
with gateway.scenario() as scenario:
scenario_id: Final = f"per-second-{pricing_case}-{uuid.uuid4().hex}"
key: Final = scenario.key()
model: Final = scenario.model(
model=f"openai/integration-per-second-{uuid.uuid4().hex}",
api_key=scenario_id,
api_base=f"{gateway.upstream_url.rstrip('/')}/v1",
**pricing,
)
with httpx.Client(base_url=gateway.upstream_url, trust_env=False) as upstream:
_clear_observations(upstream)
response: Final = gateway.request(
"POST",
"/v1/chat/completions",
{"model": model, "messages": [{"role": "user", "content": "price this request"}]},
key=key,
)
body: Final = _observed_request_body(upstream)
assert response.status_code == 200, f"{pricing_case}: {response.text}"
response_cost: Final = float(response.headers.get("x-litellm-response-cost", "0"))
duration_ms: Final = float(response.headers.get("x-litellm-response-duration-ms", "0"))
assert response_cost > 0, f"{pricing_case}: cost={response_cost}, duration_ms={duration_ms}, body={body}"
assert response_cost == pytest.approx(RATE * duration_ms / 1000, rel=1e-3), (
f"{pricing_case}: cost={response_cost}, duration_ms={duration_ms}, body={body}"
)
assert not PRICING_FIELDS.intersection(body), body
request_id: Final = string_value(object_value(response.json())["id"])
rows: Final = eventually(
lambda: read_rows(
'SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id = %s',
(request_id,),
),
lambda values: len(values) == 1,
seconds=70,
)
assert float(str(rows[0]["spend"])) == pytest.approx(response_cost, rel=1e-3)
@pytest.mark.parametrize(
("pricing_case", "pricing"),
PER_SECOND_CONFIGURATIONS,
ids=("new_field", "legacy_input", "legacy_output", "legacy_both", "all_three"),
)
def test_streaming_chat_per_second_pricing_covers_the_full_stream(
gateway: Gateway, pricing_case: str, pricing: Mapping[str, JsonValue]
) -> None:
with gateway.scenario() as scenario:
scenario_id: Final = f"per-second-stream-{pricing_case}-{uuid.uuid4().hex}"
frames: Final = _sse_frames()
handle: Final = register_scenario(
scenario_id,
SseResponse(content_type="text/event-stream", frames=frames, frame_delay_ms=FRAME_DELAY_MS),
)
scenario.cleanups.callback(delete_scenario, handle)
key: Final = scenario.key()
model: Final = scenario.model(
model=f"openai/integration-per-second-{uuid.uuid4().hex}",
api_key=scenario_id,
api_base=handle.api_base(),
**pricing,
)
with httpx.Client(base_url=gateway.upstream_url, trust_env=False) as upstream:
_clear_observations(upstream)
with gateway.client.stream(
"POST",
"/v1/chat/completions",
json={
"model": model,
"messages": [{"role": "user", "content": "price this streamed request"}],
"stream": True,
"stream_options": {"include_usage": True},
},
headers={"Authorization": f"Bearer {key}"},
) as response:
stream_lines: Final = tuple(response.iter_lines())
assert response.status_code == 200, "\n".join(stream_lines)
body: Final = _observed_request_body(upstream)
events: Final = tuple(
JSON_OBJECT.validate_json(line.removeprefix("data: "))
for line in stream_lines
if line.startswith("data: ") and line != "data: [DONE]"
)
assert len(events) == len(frames) - 1, events
assert "".join(_stream_content(event) for event in events) == "".join(CONTENT), events
usage: Final = object_value(events[-1]["usage"])
assert usage["total_tokens"] == 40, events[-1]
request_id: Final = string_value(events[0]["id"])
rows: Final = eventually(
lambda: read_rows(
'SELECT spend, request_duration_ms, '
'CAST(EXTRACT(EPOCH FROM ("endTime" - "startTime")) * 1000 AS DOUBLE PRECISION) '
'AS elapsed_duration_ms '
'FROM "LiteLLM_SpendLogs" WHERE request_id = %s',
(request_id,),
),
lambda values: len(values) == 1,
seconds=70,
)
spend: Final = float(str(rows[0]["spend"]))
request_duration_ms: Final = float(str(rows[0]["request_duration_ms"]))
elapsed_duration_ms: Final = float(str(rows[0]["elapsed_duration_ms"]))
assert spend == pytest.approx(RATE * request_duration_ms / 1000, rel=5e-2), (
f"spend={spend}, request_duration_ms={request_duration_ms}, "
f"endTime-startTime duration_ms={elapsed_duration_ms}, body={body}"
)
total_frame_delay_seconds: Final = (len(frames) - 1) * FRAME_DELAY_MS / 1000
assert spend >= RATE * total_frame_delay_seconds * 0.95, (
f"spend={spend}, total frame delay={total_frame_delay_seconds}s, body={body}"
)
assert not PRICING_FIELDS.intersection(body), body

View file

@ -713,7 +713,7 @@ def test_sagemaker_embeddings():
response = litellm.embedding(
model="sagemaker/berri-benchmarking-gpt-j-6b-fp16",
input=["good morning from litellm", "this is another item"],
input_cost_per_second=0.000420,
cost_per_second=0.000420,
)
print(f"response: {response}")
cost = completion_cost(completion_response=response)
@ -731,7 +731,7 @@ async def test_sagemaker_aembeddings():
response = await litellm.aembedding(
model="sagemaker/berri-benchmarking-gpt-j-6b-fp16",
input=["good morning from litellm", "this is another item"],
input_cost_per_second=0.000420,
cost_per_second=0.000420,
)
print(f"response: {response}")
cost = completion_cost(completion_response=response)

View file

@ -55,7 +55,7 @@ async def test_completion_sagemaker(sync_mode):
],
temperature=0.2,
max_tokens=80,
input_cost_per_second=0.000420,
cost_per_second=0.000420,
)
else:
response = await litellm.acompletion(
@ -65,7 +65,7 @@ async def test_completion_sagemaker(sync_mode):
],
temperature=0.2,
max_tokens=80,
input_cost_per_second=0.000420,
cost_per_second=0.000420,
)
# Add any assertions here to check the response
print(response)
@ -169,7 +169,7 @@ async def test_completion_sagemaker_stream(sync_mode, model):
temperature=0.2,
stream=True,
max_tokens=80,
input_cost_per_second=0.000420,
cost_per_second=0.000420,
)
for idx, chunk in enumerate(response):
@ -187,7 +187,7 @@ async def test_completion_sagemaker_stream(sync_mode, model):
stream=True,
temperature=0.2,
max_tokens=80,
input_cost_per_second=0.000420,
cost_per_second=0.000420,
)
print("streaming response")
@ -280,7 +280,7 @@ async def test_acompletion_sagemaker_non_stream():
],
temperature=0.2,
max_tokens=80,
input_cost_per_second=0.000420,
cost_per_second=0.000420,
)
# Print what was called on the mock
@ -340,7 +340,7 @@ async def test_completion_sagemaker_non_stream():
],
temperature=0.2,
max_tokens=80,
input_cost_per_second=0.000420,
cost_per_second=0.000420,
)
# Print what was called on the mock
@ -457,7 +457,7 @@ async def test_completion_sagemaker_non_stream_with_aws_params():
],
temperature=0.2,
max_tokens=80,
input_cost_per_second=0.000420,
cost_per_second=0.000420,
aws_access_key_id="gm",
aws_secret_access_key="s",
aws_region_name="us-west-5",

View file

@ -8697,7 +8697,7 @@ def test_model_has_no_cost_mapping_non_token_price_from_litellm_params_is_false(
assert model_has_no_cost_mapping(model="custom-tts", llm_router=router) is False
@pytest.mark.parametrize("cost_field", ["input_cost_per_second", "input_cost_per_token"])
@pytest.mark.parametrize("cost_field", ["cost_per_second", "input_cost_per_second", "input_cost_per_token"])
def test_model_has_no_cost_mapping_explicit_zero_price_is_false(cost_field):
from litellm.proxy.auth.auth_checks import model_has_no_cost_mapping
from litellm.router import Router

View file

@ -65,6 +65,7 @@ class TestStripClientPricingOverrides:
for field in (
"input_cost_per_token",
"output_cost_per_token",
"cost_per_second",
"input_cost_per_second",
"cache_creation_input_token_cost",
):

View file

@ -11,7 +11,7 @@ from litellm.litellm_core_utils.llm_cost_calc.zero_cost_diagnostic import (
)
from litellm.types.utils import CompletionTokensDetailsWrapper, PromptTokensDetailsWrapper, Usage
PER_SECOND_ENTRY: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042}
PER_SECOND_ENTRY: Final = {"cost_per_second": 0.00042}
FREE_ENTRY: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0, "cache_read_input_token_cost": 2e-08}
PRICED_ENTRY: Final = {"input_cost_per_token": 1e-06, "output_cost_per_token": 2e-06}
TEXT_USAGE: Final = Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)

View file

@ -607,8 +607,7 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_
litellm.register_model(
model_cost={
deployment_id: {
"input_cost_per_second": 0.02,
"output_cost_per_second": 0.04,
"cost_per_second": 0.02,
"litellm_provider": "openai",
"mode": "chat",
}
@ -627,8 +626,7 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_
logging_obj.update_environment_variables(
model="gpt-5.4-nano",
litellm_params={
"input_cost_per_second": 0.02,
"output_cost_per_second": 0.04,
"cost_per_second": 0.02,
"metadata": {"model_info": {"id": deployment_id}},
},
optional_params={},
@ -650,4 +648,4 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_
)
assert result._response_ms == pytest.approx(2000)
assert result._hidden_params["response_cost"] == pytest.approx((0.02 + 0.04) * 2)
assert result._hidden_params["response_cost"] == pytest.approx(0.02 * 2)

View file

@ -21,7 +21,13 @@ from litellm.litellm_core_utils.get_litellm_params import (
from litellm.types.litellm_params import ControlOptions
NAMED_PRICE_PARAMS: Final = frozenset(
{"input_cost_per_token", "output_cost_per_token", "input_cost_per_second", "output_cost_per_second"}
{
"input_cost_per_token",
"output_cost_per_token",
"cost_per_second",
"input_cost_per_second",
"output_cost_per_second",
}
)

View file

@ -495,7 +495,7 @@ class TestZeroCostDiagnostic:
DEPLOYMENT_ID: Final = "lit7898-query-only-priced-deployment"
MODEL_GROUP: Final = "query-only-priced-chat"
QUERY_ONLY_PRICING: Final = {"input_cost_per_query": 0.00042}
PER_SECOND_PRICING: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042}
PER_SECOND_PRICING: Final = {"cost_per_second": 0.00042}
FREE_PRICING: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0}
@pytest.fixture(params=["query_only", "free"])
@ -845,7 +845,7 @@ class TestZeroCostDiagnostic:
response: Final = self._response(usage)
response._response_ms = 1000.0
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
assert logging_obj._response_cost_calculator(result=response) == pytest.approx(0.00084)
assert logging_obj._response_cost_calculator(result=response) == pytest.approx(0.00042)
assert logging_obj.model_call_details["zero_cost_diagnostic"] is None
assert self._zero_cost_warnings(caplog) == []

View file

@ -5043,6 +5043,7 @@ class TestRouterPreRoutingAliasOverrides:
"model": "auto_router/complexity_router",
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
"cost_per_second": 0.0,
"input_cost_per_second": 0.0,
"drop_params": True,
"complexity_router_config": {"tiers": {"SIMPLE": "gpt-4o-mini"}},
@ -5064,7 +5065,12 @@ class TestRouterPreRoutingAliasOverrides:
assert result is not None
# Non-pricing alias params still carry over.
assert request_kwargs["drop_params"] is True
for field in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second"):
for field in (
"input_cost_per_token",
"output_cost_per_token",
"cost_per_second",
"input_cost_per_second",
):
assert field not in request_kwargs
@pytest.mark.asyncio

View file

@ -3037,9 +3037,9 @@ def test_completion_cost_logs_cache_and_reasoning_breakdown_for_custom_pricing()
@pytest.mark.parametrize("custom_llm_provider", ["together_ai", "openai", "anthropic", "bedrock", "azure"])
def test_cost_per_token_per_second_pricing(monkeypatch, custom_llm_provider: str):
"""
Models priced by duration (input/output_cost_per_second) with no per-token rates
Models priced by input/output duration rates with no per-token rates
must be billed as cost_per_second * response_time_ms / 1000 in cost_per_token,
whether or not the provider has its own cost calculator.
using only the input rate even when both are set, whether or not the provider has its own calculator.
"""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
@ -3064,11 +3064,40 @@ def test_cost_per_token_per_second_pricing(monkeypatch, custom_llm_provider: str
response_time_ms=1500.0,
)
assert prompt_cost == pytest.approx(0.02 * 1.5)
assert completion_cost_value == pytest.approx(0.04 * 1.5)
assert (prompt_cost, completion_cost_value) == pytest.approx((0.02 * 1.5, 0.0))
def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(monkeypatch):
def test_azure_chat_uses_token_rates_when_output_cost_per_second_is_set(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
model: Final = "test-azure-chat-token-and-output-second-pricing"
litellm.register_model(
model_cost={
model: {
"input_cost_per_token": 1e-6,
"output_cost_per_token": 2e-6,
"output_cost_per_second": 0.4,
"litellm_provider": "azure",
"mode": "chat",
}
}
)
cost: Final = cost_per_token(
model=model,
custom_llm_provider="azure",
prompt_tokens=10,
completion_tokens=20,
response_time_ms=1500.0,
)
assert cost == pytest.approx((10 * 1e-6, 20 * 2e-6))
def test_cost_per_token_ignores_cost_per_second_when_token_pricing_is_set(monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
@ -3078,8 +3107,7 @@ def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(m
model: {
"input_cost_per_token": 1e-6,
"output_cost_per_token": 2e-6,
"input_cost_per_second": 0.02,
"output_cost_per_second": 0.04,
"cost_per_second": 0.02,
"litellm_provider": "openai",
"mode": "chat",
}
@ -3098,6 +3126,39 @@ def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(m
assert completion_cost_value == pytest.approx(20 * 2e-6)
@pytest.mark.parametrize(
("pricing_fields", "expected_rate"),
[
({"cost_per_second": 0.02}, 0.02),
({"output_cost_per_second": 0.04}, 0.04),
(
{"cost_per_second": 0.05, "input_cost_per_second": 0.02, "output_cost_per_second": 0.04},
0.05,
),
({"input_cost_per_second": 0.02}, 0.02),
],
)
def test_cost_per_token_resolves_per_second_rate_precedence(
monkeypatch, pricing_fields: dict[str, float], expected_rate: float
):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
model: Final = "test-chat-per-second-rate-precedence"
entry: Final = {**pricing_fields, "litellm_provider": "together_ai", "mode": "chat"}
litellm.register_model(
model_cost={model: entry}
)
assert cost_per_token(
model=model,
custom_llm_provider="together_ai",
prompt_tokens=10,
completion_tokens=20,
response_time_ms=1500.0,
) == pytest.approx((expected_rate * 1.5, 0.0))
def _logging_obj_with_call_window(duration_ms: float) -> Logging:
start_time: Final = datetime.datetime(2026, 9, 21, 12, 0, 0)
logging_obj: Final = Logging(
@ -3160,7 +3221,7 @@ def test_completion_cost_per_second_deployment_bills_the_call_duration(
litellm_logging_obj=_logging_obj_with_call_window(logged_duration_ms),
)
assert cost == pytest.approx((0.02 + 0.04) * expected_seconds)
assert cost == pytest.approx(0.02 * expected_seconds)
@pytest.mark.parametrize("mode", ["audio_transcription", "audio_speech", "video_generation", "realtime"])

View file

@ -11,6 +11,7 @@ calculations for DB-sourced models with prompt caching pricing.
import copy
import os
from typing import Final
import pytest
@ -993,3 +994,21 @@ def test_completion_cost_applies_off_peak_only_deployment_pricing():
finally:
_restore_model_cost_entries(original_entries)
del router
def test_completion_registers_cost_per_second_pricing():
model_key: Final = "openai/test-cost-per-second-registration"
original_entries: Final = _snapshot_model_cost_entries([model_key])
try:
litellm.completion(
model=model_key,
messages=[{"role": "user", "content": "hello"}],
api_key="fake-key",
cost_per_second=0.02,
mock_response="hello back",
)
assert litellm.model_cost[model_key]["cost_per_second"] == 0.02
finally:
_restore_model_cost_entries(original_entries)

View file

@ -648,6 +648,7 @@ def validate_model_cost_values(model_data, exceptions=None):
"output_cost_per_image_4K",
"input_cost_per_pixel",
"output_cost_per_pixel",
"cost_per_second",
"input_cost_per_second",
"output_cost_per_second",
"output_cost_per_second_480p",
@ -829,6 +830,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"input_cost_per_pixel": {"type": "number"},
"input_cost_per_query": {"type": "number"},
"input_cost_per_request": {"type": "number"},
"cost_per_second": {"type": "number"},
"input_cost_per_second": {"type": "number"},
"input_cost_per_token": {"type": "number"},
"input_cost_per_token_above_128k_tokens": {"type": "number"},

View file

@ -40,6 +40,7 @@ def test_custom_pricing_params_keeps_every_field_it_had():
"output_cost_per_character",
"cache_read_input_token_cost",
"cache_creation_input_token_cost",
"cost_per_second",
"input_cost_per_second",
"cache_read_input_token_cost_flex",
"input_cost_per_character_above_128k_tokens",

View file

@ -33013,6 +33013,8 @@ export interface components {
complexity_router_default_model?: string | null;
/** Configurable Clientside Auth Params */
configurable_clientside_auth_params?: (string | components["schemas"]["ConfigurableClientsideParamsCustomAuth-Input"])[] | null;
/** Cost Per Second */
cost_per_second?: number | null;
/** Custom Llm Provider */
custom_llm_provider?: string | null;
/** Default Api Key Rpm Limit */
@ -46842,6 +46844,8 @@ export interface components {
complexity_router_default_model?: string | null;
/** Configurable Clientside Auth Params */
configurable_clientside_auth_params?: (string | components["schemas"]["ConfigurableClientsideParamsCustomAuth-Input"])[] | null;
/** Cost Per Second */
cost_per_second?: number | null;
/** Custom Llm Provider */
custom_llm_provider?: string | null;
/** Default Api Key Rpm Limit */