diff --git a/cookbook/misc/config.yaml b/cookbook/misc/config.yaml index 27a6332a882..a485bf825fc 100644 --- a/cookbook/misc/config.yaml +++ b/cookbook/misc/config.yaml @@ -24,7 +24,7 @@ model_list: - model_name: sagemaker-completion-model litellm_params: model: sagemaker/berri-benchmarking-Llama-2-70b-chat-hf-4 - input_cost_per_second: 0.000420 + cost_per_second: 0.000420 - model_name: text-embedding-ada-002 litellm_params: model: azure/azure-embedding-model diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index aa543439885..75f16e0c00d 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -125,6 +125,8 @@ pub struct ModelInfo { pub computer_use_input_cost_per_1k_tokens: Option, #[serde(skip_serializing_if = "Option::is_none")] pub computer_use_output_cost_per_1k_tokens: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub cost_per_second: Option, /// Reasoning effort the provider applies when the request omits reasoning_effort. Gates whether a non-default temperature or the top_p/logprobs sampling params are accepted, which hold only when the effort resolves to 'none'. #[serde(skip_serializing_if = "Option::is_none")] pub default_reasoning_effort: Option, diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index a279b9f0903..2f76b84f5e3 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -351,19 +351,27 @@ def _per_second_pricing_cost( return None if _has_token_or_tiered_pricing(model_info) or not _bills_wall_clock_seconds(model_info): return None + cost_per_second: Final = model_info.get("cost_per_second") input_cost_per_second: Final = model_info.get("input_cost_per_second") output_cost_per_second: Final = model_info.get("output_cost_per_second") - if input_cost_per_second is None and output_cost_per_second is None: + resolved_cost_per_second: Final = ( + cost_per_second + if cost_per_second is not None + else input_cost_per_second + if input_cost_per_second is not None + else output_cost_per_second + ) + if resolved_cost_per_second is None: return None + seconds: Final = (response_time_ms or 0.0) / 1000 verbose_logger.debug( - "For model=%s - input_cost_per_second: %s; output_cost_per_second: %s; response time: %s", + "For model=%s - cost_per_second: %s; response time: %s", model, - input_cost_per_second, - output_cost_per_second, + resolved_cost_per_second, response_time_ms, ) - return (input_cost_per_second or 0.0) * seconds, (output_cost_per_second or 0.0) * seconds + return resolved_cost_per_second * seconds, 0.0 def cost_per_token( @@ -790,7 +798,9 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None return value if isinstance(value, str) and value else None -_NON_TOKEN_RATE_FIELDS: Final = frozenset({"input_cost_per_second", "input_cost_per_query", "tiered_pricing"}) +_NON_TOKEN_RATE_FIELDS: Final = frozenset( + {"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"} +) def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool: diff --git a/litellm/litellm_core_utils/get_litellm_params.py b/litellm/litellm_core_utils/get_litellm_params.py index f28259a1b7f..b8441d2bc6d 100644 --- a/litellm/litellm_core_utils/get_litellm_params.py +++ b/litellm/litellm_core_utils/get_litellm_params.py @@ -155,6 +155,7 @@ def get_litellm_params( allm_passthrough_route=None, preset_cache_key=None, no_log=None, + cost_per_second: float | None = None, input_cost_per_second=None, input_cost_per_token=None, output_cost_per_token=None, @@ -216,6 +217,7 @@ def get_litellm_params( "preset_cache_key": preset_cache_key, "no-log": no_log or kwargs.get("no-log"), "stream_response": {}, # litellm_call_id: ModelResponse Dict + "cost_per_second": cost_per_second, "input_cost_per_token": input_cost_per_token, "input_cost_per_second": input_cost_per_second, "output_cost_per_token": output_cost_per_token, diff --git a/litellm/llms/azure/cost_calculation.py b/litellm/llms/azure/cost_calculation.py index 8dc809507d5..057e9dbb9d9 100644 --- a/litellm/llms/azure/cost_calculation.py +++ b/litellm/llms/azure/cost_calculation.py @@ -3,12 +3,8 @@ Helper util for handling azure openai-specific cost calculation - e.g.: prompt caching, audio tokens """ -from typing import Final - -from litellm._logging import verbose_logger from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token from litellm.types.utils import Usage -from litellm.utils import get_model_info def cost_per_token( @@ -27,26 +23,6 @@ def cost_per_token( Returns: Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd """ - ## GET MODEL INFO - model_info: Final = get_model_info(model=model, custom_llm_provider="azure") - - ## Speech / Audio cost calculation (cost per second for TTS models) - if ( - "output_cost_per_second" in model_info - and model_info["output_cost_per_second"] is not None - and response_time_ms is not None - ): - verbose_logger.debug( - "For model=%s - output_cost_per_second: %s; response time: %s", - model, - model_info.get("output_cost_per_second"), - response_time_ms, - ) - ## COST PER SECOND ## - prompt_cost: Final = 0.0 - completion_cost: Final = model_info["output_cost_per_second"] * response_time_ms / 1000 - return prompt_cost, completion_cost - ## Use generic cost calculator for all other cases ## This properly handles: text tokens, audio tokens, cached tokens, reasoning tokens, etc. return generic_cost_per_token( diff --git a/litellm/main.py b/litellm/main.py index 6c85adf3ae8..769eac79488 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -5353,6 +5353,7 @@ def completion( ### CUSTOM MODEL COST ### input_cost_per_token: Final = kwargs.get("input_cost_per_token", None) output_cost_per_token: Final = kwargs.get("output_cost_per_token", None) + cost_per_second: Final = kwargs.get("cost_per_second", None) input_cost_per_second: Final = kwargs.get("input_cost_per_second", None) output_cost_per_second: Final = kwargs.get("output_cost_per_second", None) ### CUSTOM PROMPT TEMPLATE ### @@ -5514,8 +5515,11 @@ def completion( ### REGISTER CUSTOM MODEL PRICING -- IF GIVEN ### if ( - input_cost_per_token is not None and output_cost_per_token is not None - ) or input_cost_per_second is not None: + (input_cost_per_token is not None and output_cost_per_token is not None) + or input_cost_per_second is not None + or output_cost_per_second is not None + or cost_per_second is not None + ): _register_custom_pricing_for_request( model=model, custom_llm_provider=custom_llm_provider, @@ -5657,6 +5661,7 @@ def completion( proxy_server_request=proxy_server_request, preset_cache_key=preset_cache_key, no_log=no_log, + cost_per_second=cost_per_second, input_cost_per_second=input_cost_per_second, input_cost_per_token=input_cost_per_token, output_cost_per_second=output_cost_per_second, @@ -6354,7 +6359,9 @@ def embedding( ### CUSTOM MODEL COST ### input_cost_per_token: Final = kwargs.get("input_cost_per_token", None) output_cost_per_token: Final = kwargs.get("output_cost_per_token", None) + cost_per_second: Final = kwargs.get("cost_per_second", None) input_cost_per_second: Final = kwargs.get("input_cost_per_second", None) + output_cost_per_second: Final = kwargs.get("output_cost_per_second", None) openai_params: Final = [ "user", "dimensions", @@ -6395,7 +6402,12 @@ def embedding( ) ### REGISTER CUSTOM MODEL PRICING -- IF GIVEN ### - if (input_cost_per_token is not None and output_cost_per_token is not None) or input_cost_per_second is not None: + if ( + (input_cost_per_token is not None and output_cost_per_token is not None) + or input_cost_per_second is not None + or output_cost_per_second is not None + or cost_per_second is not None + ): _register_custom_pricing_for_request( model=model, custom_llm_provider=custom_llm_provider, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c49c012cc78..9ae270243d2 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -12682,43 +12682,43 @@ "source": "https://developers.openai.com/api/docs/pricing" }, "bedrock/*/1-month-commitment/cohere.command-light-text-v14": { + "cost_per_second": 0.001902, "input_cost_per_second": 0.001902, "litellm_provider": "bedrock", "max_input_tokens": 4096, "max_output_tokens": 4096, "max_tokens": 4096, "mode": "chat", - "output_cost_per_second": 0.001902, "supports_tool_choice": true }, "bedrock/*/1-month-commitment/cohere.command-text-v14": { + "cost_per_second": 0.011, "input_cost_per_second": 0.011, "litellm_provider": "bedrock", "max_input_tokens": 4096, "max_output_tokens": 4096, "max_tokens": 4096, "mode": "chat", - "output_cost_per_second": 0.011, "supports_tool_choice": true }, "bedrock/*/6-month-commitment/cohere.command-light-text-v14": { + "cost_per_second": 0.0011416, "input_cost_per_second": 0.0011416, "litellm_provider": "bedrock", "max_input_tokens": 4096, "max_output_tokens": 4096, "max_tokens": 4096, "mode": "chat", - "output_cost_per_second": 0.0011416, "supports_tool_choice": true }, "bedrock/*/6-month-commitment/cohere.command-text-v14": { + "cost_per_second": 0.0066027, "input_cost_per_second": 0.0066027, "litellm_provider": "bedrock", "max_input_tokens": 4096, "max_output_tokens": 4096, "max_tokens": 4096, "mode": "chat", - "output_cost_per_second": 0.0066027, "supports_tool_choice": true }, "bedrock/guardrails": { @@ -12737,61 +12737,61 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.01475, "input_cost_per_second": 0.01475, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.01475, "supports_tool_choice": true }, "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.0455, "input_cost_per_second": 0.0455, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.0455 + "mode": "chat" }, "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.0455, "input_cost_per_second": 0.0455, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.0455, "supports_tool_choice": true }, "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.008194, "input_cost_per_second": 0.008194, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.008194, "supports_tool_choice": true }, "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.02527, "input_cost_per_second": 0.02527, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.02527 + "mode": "chat" }, "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.02527, "input_cost_per_second": 0.02527, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.02527, "supports_tool_choice": true }, "bedrock/ap-northeast-1/anthropic.claude-instant-v1": { @@ -13241,61 +13241,61 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.01635, "input_cost_per_second": 0.01635, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.01635, "supports_tool_choice": true }, "bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.0415, "input_cost_per_second": 0.0415, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.0415 + "mode": "chat" }, "bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.0415, "input_cost_per_second": 0.0415, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.0415, "supports_tool_choice": true }, "bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.009083, "input_cost_per_second": 0.009083, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.009083, "supports_tool_choice": true }, "bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.02305, "input_cost_per_second": 0.02305, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.02305 + "mode": "chat" }, "bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.02305, "input_cost_per_second": 0.02305, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.02305, "supports_tool_choice": true }, "bedrock/eu-central-1/anthropic.claude-instant-v1": { @@ -13737,61 +13737,61 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.011, "input_cost_per_second": 0.011, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.011, "supports_tool_choice": true }, "bedrock/us-east-1/1-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.0175, "input_cost_per_second": 0.0175, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.0175 + "mode": "chat" }, "bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.0175, "input_cost_per_second": 0.0175, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.0175, "supports_tool_choice": true }, "bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.00611, "input_cost_per_second": 0.00611, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.00611, "supports_tool_choice": true }, "bedrock/us-east-1/6-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.00972, "input_cost_per_second": 0.00972, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.00972 + "mode": "chat" }, "bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.00972, "input_cost_per_second": 0.00972, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.00972, "supports_tool_choice": true }, "bedrock/us-east-1/anthropic.claude-instant-v1": { @@ -14385,61 +14385,61 @@ "output_cost_per_token": 6e-07 }, "bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.011, "input_cost_per_second": 0.011, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.011, "supports_tool_choice": true }, "bedrock/us-west-2/1-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.0175, "input_cost_per_second": 0.0175, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.0175 + "mode": "chat" }, "bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.0175, "input_cost_per_second": 0.0175, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.0175, "supports_tool_choice": true }, "bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.00611, "input_cost_per_second": 0.00611, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.00611, "supports_tool_choice": true }, "bedrock/us-west-2/6-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.00972, "input_cost_per_second": 0.00972, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.00972 + "mode": "chat" }, "bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.00972, "input_cost_per_second": 0.00972, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.00972, "supports_tool_choice": true }, "bedrock/us-west-2/anthropic.claude-instant-v1": { @@ -38527,7 +38527,6 @@ }, "mistral/voxtral-small-2507": { "cache_read_input_token_cost": 1e-08, - "input_cost_per_second": 6.666666666666667e-05, "input_cost_per_token": 1e-07, "litellm_provider": "mistral", "max_input_tokens": 32768, @@ -38543,7 +38542,6 @@ }, "mistral/voxtral-small-latest": { "cache_read_input_token_cost": 1e-08, - "input_cost_per_second": 6.666666666666667e-05, "input_cost_per_token": 1e-07, "litellm_provider": "mistral", "max_input_tokens": 32768, diff --git a/litellm/router.py b/litellm/router.py index 86ba5112435..842cd9de378 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8801,7 +8801,7 @@ class Router: return if any( model_info.get(field) is not None - for field in ("input_cost_per_token", "input_cost_per_second", "tiered_pricing") + for field in ("input_cost_per_token", "input_cost_per_second", "cost_per_second", "tiered_pricing") ): return try: diff --git a/litellm/types/router.py b/litellm/types/router.py index d545f7ae639..2ab1a1185ed 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -606,6 +606,7 @@ class LiteLLMParamsTypedDict(TypedDict, total=False): ## CUSTOM PRICING ## input_cost_per_token: float | None output_cost_per_token: float | None + cost_per_second: ReadOnly[float | None] input_cost_per_second: float | None output_cost_per_second: float | None output_cost_per_second_480p: ReadOnly[float | None] diff --git a/litellm/types/utils.py b/litellm/types/utils.py index dba15bc99a5..e32e3b74ec6 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -329,6 +329,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): input_cost_per_video_per_second: float | None # only for vertex ai models input_cost_per_audio_token_batches: ReadOnly[float | None] input_cost_per_image_token_batches: ReadOnly[float | None] + cost_per_second: ReadOnly[float | None] input_cost_per_second: float | None # for OpenAI Speech models input_cost_per_token_batches: float | None input_cost_per_video_token_batches: ReadOnly[float | None] @@ -2784,6 +2785,7 @@ class LoggedLiteLLMParams(TypedDict, total=False): acompletion: bool | None preset_cache_key: str | None no_log: bool | None + cost_per_second: ReadOnly[float | None] input_cost_per_second: float | None input_cost_per_token: float | None output_cost_per_token: float | None @@ -3709,6 +3711,7 @@ class MirroredPricingParams(BaseModel): class CustomPricingLiteLLMParams(MirroredPricingParams): ## CUSTOM PRICING ## + cost_per_second: float | None = None input_cost_per_second: float | None = None output_cost_per_second: float | None = None output_cost_per_second_1080p: float | None = None diff --git a/litellm/utils.py b/litellm/utils.py index 09b5067339d..ffd507fad45 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -6168,6 +6168,7 @@ def _get_model_info_helper( ), input_cost_per_token_above_512k_tokens=_model_info.get("input_cost_per_token_above_512k_tokens", None), input_cost_per_query=_model_info.get("input_cost_per_query", None), + cost_per_second=_model_info.get("cost_per_second", None), input_cost_per_second=_model_info.get("input_cost_per_second", None), input_cost_per_audio_token=_model_info.get("input_cost_per_audio_token", None), input_cost_per_image_token=_model_info.get("input_cost_per_image_token", None), diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c49c012cc78..9ae270243d2 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -12682,43 +12682,43 @@ "source": "https://developers.openai.com/api/docs/pricing" }, "bedrock/*/1-month-commitment/cohere.command-light-text-v14": { + "cost_per_second": 0.001902, "input_cost_per_second": 0.001902, "litellm_provider": "bedrock", "max_input_tokens": 4096, "max_output_tokens": 4096, "max_tokens": 4096, "mode": "chat", - "output_cost_per_second": 0.001902, "supports_tool_choice": true }, "bedrock/*/1-month-commitment/cohere.command-text-v14": { + "cost_per_second": 0.011, "input_cost_per_second": 0.011, "litellm_provider": "bedrock", "max_input_tokens": 4096, "max_output_tokens": 4096, "max_tokens": 4096, "mode": "chat", - "output_cost_per_second": 0.011, "supports_tool_choice": true }, "bedrock/*/6-month-commitment/cohere.command-light-text-v14": { + "cost_per_second": 0.0011416, "input_cost_per_second": 0.0011416, "litellm_provider": "bedrock", "max_input_tokens": 4096, "max_output_tokens": 4096, "max_tokens": 4096, "mode": "chat", - "output_cost_per_second": 0.0011416, "supports_tool_choice": true }, "bedrock/*/6-month-commitment/cohere.command-text-v14": { + "cost_per_second": 0.0066027, "input_cost_per_second": 0.0066027, "litellm_provider": "bedrock", "max_input_tokens": 4096, "max_output_tokens": 4096, "max_tokens": 4096, "mode": "chat", - "output_cost_per_second": 0.0066027, "supports_tool_choice": true }, "bedrock/guardrails": { @@ -12737,61 +12737,61 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.01475, "input_cost_per_second": 0.01475, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.01475, "supports_tool_choice": true }, "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.0455, "input_cost_per_second": 0.0455, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.0455 + "mode": "chat" }, "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.0455, "input_cost_per_second": 0.0455, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.0455, "supports_tool_choice": true }, "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.008194, "input_cost_per_second": 0.008194, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.008194, "supports_tool_choice": true }, "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.02527, "input_cost_per_second": 0.02527, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.02527 + "mode": "chat" }, "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.02527, "input_cost_per_second": 0.02527, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.02527, "supports_tool_choice": true }, "bedrock/ap-northeast-1/anthropic.claude-instant-v1": { @@ -13241,61 +13241,61 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.01635, "input_cost_per_second": 0.01635, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.01635, "supports_tool_choice": true }, "bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.0415, "input_cost_per_second": 0.0415, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.0415 + "mode": "chat" }, "bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.0415, "input_cost_per_second": 0.0415, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.0415, "supports_tool_choice": true }, "bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.009083, "input_cost_per_second": 0.009083, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.009083, "supports_tool_choice": true }, "bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.02305, "input_cost_per_second": 0.02305, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.02305 + "mode": "chat" }, "bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.02305, "input_cost_per_second": 0.02305, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.02305, "supports_tool_choice": true }, "bedrock/eu-central-1/anthropic.claude-instant-v1": { @@ -13737,61 +13737,61 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.011, "input_cost_per_second": 0.011, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.011, "supports_tool_choice": true }, "bedrock/us-east-1/1-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.0175, "input_cost_per_second": 0.0175, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.0175 + "mode": "chat" }, "bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.0175, "input_cost_per_second": 0.0175, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.0175, "supports_tool_choice": true }, "bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.00611, "input_cost_per_second": 0.00611, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.00611, "supports_tool_choice": true }, "bedrock/us-east-1/6-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.00972, "input_cost_per_second": 0.00972, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.00972 + "mode": "chat" }, "bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.00972, "input_cost_per_second": 0.00972, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.00972, "supports_tool_choice": true }, "bedrock/us-east-1/anthropic.claude-instant-v1": { @@ -14385,61 +14385,61 @@ "output_cost_per_token": 6e-07 }, "bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.011, "input_cost_per_second": 0.011, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.011, "supports_tool_choice": true }, "bedrock/us-west-2/1-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.0175, "input_cost_per_second": 0.0175, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.0175 + "mode": "chat" }, "bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.0175, "input_cost_per_second": 0.0175, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.0175, "supports_tool_choice": true }, "bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1": { + "cost_per_second": 0.00611, "input_cost_per_second": 0.00611, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.00611, "supports_tool_choice": true }, "bedrock/us-west-2/6-month-commitment/anthropic.claude-v1": { + "cost_per_second": 0.00972, "input_cost_per_second": 0.00972, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, - "mode": "chat", - "output_cost_per_second": 0.00972 + "mode": "chat" }, "bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1": { + "cost_per_second": 0.00972, "input_cost_per_second": 0.00972, "litellm_provider": "bedrock", "max_input_tokens": 100000, "max_output_tokens": 8191, "max_tokens": 8191, "mode": "chat", - "output_cost_per_second": 0.00972, "supports_tool_choice": true }, "bedrock/us-west-2/anthropic.claude-instant-v1": { @@ -38527,7 +38527,6 @@ }, "mistral/voxtral-small-2507": { "cache_read_input_token_cost": 1e-08, - "input_cost_per_second": 6.666666666666667e-05, "input_cost_per_token": 1e-07, "litellm_provider": "mistral", "max_input_tokens": 32768, @@ -38543,7 +38542,6 @@ }, "mistral/voxtral-small-latest": { "cache_read_input_token_cost": 1e-08, - "input_cost_per_second": 6.666666666666667e-05, "input_cost_per_token": 1e-07, "litellm_provider": "mistral", "max_input_tokens": 32768, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index e893b6265fa..c4b424a26b4 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -249,6 +249,10 @@ "comment": { "type": "string" }, + "cost_per_second": { + "type": "number", + "minimum": 0 + }, "default_reasoning_effort": { "type": "string", "description": "Reasoning effort the provider applies when the request omits reasoning_effort. Gates whether a non-default temperature or the top_p/logprobs sampling params are accepted, which hold only when the effort resolves to 'none'.", diff --git a/proxy_server_config.yaml b/proxy_server_config.yaml index 24e26ea8e22..be6dd20647d 100644 --- a/proxy_server_config.yaml +++ b/proxy_server_config.yaml @@ -31,7 +31,7 @@ model_list: - model_name: sagemaker-completion-model litellm_params: model: sagemaker/berri-benchmarking-Llama-2-70b-chat-hf-4 - input_cost_per_second: 0.000420 + cost_per_second: 0.000420 - model_name: text-embedding-ada-002 litellm_params: model: openai/text-embedding-3-small diff --git a/tests/integration/pricing/test_per_second_pricing.py b/tests/integration/pricing/test_per_second_pricing.py new file mode 100644 index 00000000000..ad44a631054 --- /dev/null +++ b/tests/integration/pricing/test_per_second_pricing.py @@ -0,0 +1,196 @@ +import json +import uuid +from collections.abc import Mapping +from typing import Final + +import httpx +import pytest +from pydantic import JsonValue + +from tests.integration._support.client import JSON_OBJECT, Gateway, eventually, object_value, string_value +from tests.integration._support.database import read_rows +from tests.integration._support.upstream import delete_scenario, register_scenario +from tests.integration.cost_calculation.cost_tracking_case import SseResponse + +RATE: Final = 0.5 +FRAME_DELAY_MS: Final = 300 +CONTENT: Final = ("one", " two", " three", " four") +PRICING_FIELDS: Final = frozenset({"cost_per_second", "input_cost_per_second", "output_cost_per_second"}) +PER_SECOND_CONFIGURATIONS: Final[tuple[tuple[str, Mapping[str, JsonValue]], ...]] = ( + ("new_field", {"cost_per_second": RATE}), + ("legacy_input", {"input_cost_per_second": RATE}), + ("legacy_output", {"output_cost_per_second": RATE}), + ("legacy_both", {"input_cost_per_second": RATE, "output_cost_per_second": 0.25}), + ( + "all_three", + {"cost_per_second": RATE, "input_cost_per_second": 0.25, "output_cost_per_second": 0.125}, + ), +) + + +def _sse_chunk(delta: dict[str, JsonValue], finish_reason: str | None) -> str: + payload: Final = { + "id": "$REQUEST_ID", + "object": "chat.completion.chunk", + "created": 1, + "model": "integration-per-second", + "choices": [{"index": 0, "delta": delta, "finish_reason": finish_reason}], + } + return f"data: {json.dumps(payload)}" + + +def _sse_frames() -> tuple[str, ...]: + content_frames: Final = tuple(_sse_chunk({"content": content}, None) for content in CONTENT) + usage_payload: Final = { + "id": "$REQUEST_ID", + "object": "chat.completion.chunk", + "created": 1, + "model": "integration-per-second", + "choices": [], + "usage": {"prompt_tokens": 20, "completion_tokens": 20, "total_tokens": 40}, + } + usage_frame: Final = f"data: {json.dumps(usage_payload)}" + return (*content_frames, _sse_chunk({}, "stop"), usage_frame, "data: [DONE]") + + +def _stream_content(event: dict[str, JsonValue]) -> str: + choices: Final = event.get("choices") + if not isinstance(choices, list) or not choices: + return "" + delta: Final = object_value(object_value(choices[0])["delta"]) + content: Final = delta.get("content") + return content if isinstance(content, str) else "" + + +def _clear_observations(upstream: httpx.Client) -> None: + response: Final = upstream.get("/__observations") + assert response.status_code == 200, response.text + + +def _observed_request_body(upstream: httpx.Client) -> dict[str, JsonValue]: + observations: Final = JSON_OBJECT.validate_json(upstream.get("/__observations").content)["requests"] + assert isinstance(observations, list) + assert len(observations) == 1 + return object_value(object_value(observations[0])["body"]) + + +@pytest.mark.parametrize( + ("pricing_case", "pricing"), + PER_SECOND_CONFIGURATIONS, + ids=("new_field", "legacy_input", "legacy_output", "legacy_both", "all_three"), +) +def test_chat_per_second_pricing_is_charged_once_and_not_forwarded( + gateway: Gateway, pricing_case: str, pricing: Mapping[str, JsonValue] +) -> None: + with gateway.scenario() as scenario: + scenario_id: Final = f"per-second-{pricing_case}-{uuid.uuid4().hex}" + key: Final = scenario.key() + model: Final = scenario.model( + model=f"openai/integration-per-second-{uuid.uuid4().hex}", + api_key=scenario_id, + api_base=f"{gateway.upstream_url.rstrip('/')}/v1", + **pricing, + ) + with httpx.Client(base_url=gateway.upstream_url, trust_env=False) as upstream: + _clear_observations(upstream) + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": "price this request"}]}, + key=key, + ) + body: Final = _observed_request_body(upstream) + assert response.status_code == 200, f"{pricing_case}: {response.text}" + response_cost: Final = float(response.headers.get("x-litellm-response-cost", "0")) + duration_ms: Final = float(response.headers.get("x-litellm-response-duration-ms", "0")) + assert response_cost > 0, f"{pricing_case}: cost={response_cost}, duration_ms={duration_ms}, body={body}" + assert response_cost == pytest.approx(RATE * duration_ms / 1000, rel=1e-3), ( + f"{pricing_case}: cost={response_cost}, duration_ms={duration_ms}, body={body}" + ) + assert not PRICING_FIELDS.intersection(body), body + + request_id: Final = string_value(object_value(response.json())["id"]) + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id = %s', + (request_id,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(str(rows[0]["spend"])) == pytest.approx(response_cost, rel=1e-3) + + +@pytest.mark.parametrize( + ("pricing_case", "pricing"), + PER_SECOND_CONFIGURATIONS, + ids=("new_field", "legacy_input", "legacy_output", "legacy_both", "all_three"), +) +def test_streaming_chat_per_second_pricing_covers_the_full_stream( + gateway: Gateway, pricing_case: str, pricing: Mapping[str, JsonValue] +) -> None: + with gateway.scenario() as scenario: + scenario_id: Final = f"per-second-stream-{pricing_case}-{uuid.uuid4().hex}" + frames: Final = _sse_frames() + handle: Final = register_scenario( + scenario_id, + SseResponse(content_type="text/event-stream", frames=frames, frame_delay_ms=FRAME_DELAY_MS), + ) + scenario.cleanups.callback(delete_scenario, handle) + key: Final = scenario.key() + model: Final = scenario.model( + model=f"openai/integration-per-second-{uuid.uuid4().hex}", + api_key=scenario_id, + api_base=handle.api_base(), + **pricing, + ) + with httpx.Client(base_url=gateway.upstream_url, trust_env=False) as upstream: + _clear_observations(upstream) + with gateway.client.stream( + "POST", + "/v1/chat/completions", + json={ + "model": model, + "messages": [{"role": "user", "content": "price this streamed request"}], + "stream": True, + "stream_options": {"include_usage": True}, + }, + headers={"Authorization": f"Bearer {key}"}, + ) as response: + stream_lines: Final = tuple(response.iter_lines()) + assert response.status_code == 200, "\n".join(stream_lines) + body: Final = _observed_request_body(upstream) + + events: Final = tuple( + JSON_OBJECT.validate_json(line.removeprefix("data: ")) + for line in stream_lines + if line.startswith("data: ") and line != "data: [DONE]" + ) + assert len(events) == len(frames) - 1, events + assert "".join(_stream_content(event) for event in events) == "".join(CONTENT), events + usage: Final = object_value(events[-1]["usage"]) + assert usage["total_tokens"] == 40, events[-1] + request_id: Final = string_value(events[0]["id"]) + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, request_duration_ms, ' + 'CAST(EXTRACT(EPOCH FROM ("endTime" - "startTime")) * 1000 AS DOUBLE PRECISION) ' + 'AS elapsed_duration_ms ' + 'FROM "LiteLLM_SpendLogs" WHERE request_id = %s', + (request_id,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + spend: Final = float(str(rows[0]["spend"])) + request_duration_ms: Final = float(str(rows[0]["request_duration_ms"])) + elapsed_duration_ms: Final = float(str(rows[0]["elapsed_duration_ms"])) + assert spend == pytest.approx(RATE * request_duration_ms / 1000, rel=5e-2), ( + f"spend={spend}, request_duration_ms={request_duration_ms}, " + f"endTime-startTime duration_ms={elapsed_duration_ms}, body={body}" + ) + total_frame_delay_seconds: Final = (len(frames) - 1) * FRAME_DELAY_MS / 1000 + assert spend >= RATE * total_frame_delay_seconds * 0.95, ( + f"spend={spend}, total frame delay={total_frame_delay_seconds}s, body={body}" + ) + assert not PRICING_FIELDS.intersection(body), body diff --git a/tests/local_testing/test_embedding.py b/tests/local_testing/test_embedding.py index acbc4f20405..19885f891c0 100644 --- a/tests/local_testing/test_embedding.py +++ b/tests/local_testing/test_embedding.py @@ -713,7 +713,7 @@ def test_sagemaker_embeddings(): response = litellm.embedding( model="sagemaker/berri-benchmarking-gpt-j-6b-fp16", input=["good morning from litellm", "this is another item"], - input_cost_per_second=0.000420, + cost_per_second=0.000420, ) print(f"response: {response}") cost = completion_cost(completion_response=response) @@ -731,7 +731,7 @@ async def test_sagemaker_aembeddings(): response = await litellm.aembedding( model="sagemaker/berri-benchmarking-gpt-j-6b-fp16", input=["good morning from litellm", "this is another item"], - input_cost_per_second=0.000420, + cost_per_second=0.000420, ) print(f"response: {response}") cost = completion_cost(completion_response=response) diff --git a/tests/local_testing/test_sagemaker.py b/tests/local_testing/test_sagemaker.py index a01c8c217c6..bcbe230bc0a 100644 --- a/tests/local_testing/test_sagemaker.py +++ b/tests/local_testing/test_sagemaker.py @@ -55,7 +55,7 @@ async def test_completion_sagemaker(sync_mode): ], temperature=0.2, max_tokens=80, - input_cost_per_second=0.000420, + cost_per_second=0.000420, ) else: response = await litellm.acompletion( @@ -65,7 +65,7 @@ async def test_completion_sagemaker(sync_mode): ], temperature=0.2, max_tokens=80, - input_cost_per_second=0.000420, + cost_per_second=0.000420, ) # Add any assertions here to check the response print(response) @@ -169,7 +169,7 @@ async def test_completion_sagemaker_stream(sync_mode, model): temperature=0.2, stream=True, max_tokens=80, - input_cost_per_second=0.000420, + cost_per_second=0.000420, ) for idx, chunk in enumerate(response): @@ -187,7 +187,7 @@ async def test_completion_sagemaker_stream(sync_mode, model): stream=True, temperature=0.2, max_tokens=80, - input_cost_per_second=0.000420, + cost_per_second=0.000420, ) print("streaming response") @@ -280,7 +280,7 @@ async def test_acompletion_sagemaker_non_stream(): ], temperature=0.2, max_tokens=80, - input_cost_per_second=0.000420, + cost_per_second=0.000420, ) # Print what was called on the mock @@ -340,7 +340,7 @@ async def test_completion_sagemaker_non_stream(): ], temperature=0.2, max_tokens=80, - input_cost_per_second=0.000420, + cost_per_second=0.000420, ) # Print what was called on the mock @@ -457,7 +457,7 @@ async def test_completion_sagemaker_non_stream_with_aws_params(): ], temperature=0.2, max_tokens=80, - input_cost_per_second=0.000420, + cost_per_second=0.000420, aws_access_key_id="gm", aws_secret_access_key="s", aws_region_name="us-west-5", diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py index dabd97cff0b..d3cb8e4a645 100644 --- a/tests/test_litellm/proxy/auth/test_auth_checks.py +++ b/tests/test_litellm/proxy/auth/test_auth_checks.py @@ -8697,7 +8697,7 @@ def test_model_has_no_cost_mapping_non_token_price_from_litellm_params_is_false( assert model_has_no_cost_mapping(model="custom-tts", llm_router=router) is False -@pytest.mark.parametrize("cost_field", ["input_cost_per_second", "input_cost_per_token"]) +@pytest.mark.parametrize("cost_field", ["cost_per_second", "input_cost_per_second", "input_cost_per_token"]) def test_model_has_no_cost_mapping_explicit_zero_price_is_false(cost_field): from litellm.proxy.auth.auth_checks import model_has_no_cost_mapping from litellm.router import Router diff --git a/tests/test_litellm/proxy/test_pricing_field_strip.py b/tests/test_litellm/proxy/test_pricing_field_strip.py index a84c6ba2b8a..a0e25e91f37 100644 --- a/tests/test_litellm/proxy/test_pricing_field_strip.py +++ b/tests/test_litellm/proxy/test_pricing_field_strip.py @@ -65,6 +65,7 @@ class TestStripClientPricingOverrides: for field in ( "input_cost_per_token", "output_cost_per_token", + "cost_per_second", "input_cost_per_second", "cache_creation_input_token_cost", ): diff --git a/tests/unit/litellm_core_utils/llm_cost_calc/test_zero_cost_diagnostic.py b/tests/unit/litellm_core_utils/llm_cost_calc/test_zero_cost_diagnostic.py index 0e453e3f5eb..921dde8bf7b 100644 --- a/tests/unit/litellm_core_utils/llm_cost_calc/test_zero_cost_diagnostic.py +++ b/tests/unit/litellm_core_utils/llm_cost_calc/test_zero_cost_diagnostic.py @@ -11,7 +11,7 @@ from litellm.litellm_core_utils.llm_cost_calc.zero_cost_diagnostic import ( ) from litellm.types.utils import CompletionTokensDetailsWrapper, PromptTokensDetailsWrapper, Usage -PER_SECOND_ENTRY: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042} +PER_SECOND_ENTRY: Final = {"cost_per_second": 0.00042} FREE_ENTRY: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0, "cache_read_input_token_cost": 2e-08} PRICED_ENTRY: Final = {"input_cost_per_token": 1e-06, "output_cost_per_token": 2e-06} TEXT_USAGE: Final = Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30) diff --git a/tests/unit/litellm_core_utils/llm_response_utils/test_response_metadata.py b/tests/unit/litellm_core_utils/llm_response_utils/test_response_metadata.py index 6f297e6e06a..554447f4273 100644 --- a/tests/unit/litellm_core_utils/llm_response_utils/test_response_metadata.py +++ b/tests/unit/litellm_core_utils/llm_response_utils/test_response_metadata.py @@ -607,8 +607,7 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_ litellm.register_model( model_cost={ deployment_id: { - "input_cost_per_second": 0.02, - "output_cost_per_second": 0.04, + "cost_per_second": 0.02, "litellm_provider": "openai", "mode": "chat", } @@ -627,8 +626,7 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_ logging_obj.update_environment_variables( model="gpt-5.4-nano", litellm_params={ - "input_cost_per_second": 0.02, - "output_cost_per_second": 0.04, + "cost_per_second": 0.02, "metadata": {"model_info": {"id": deployment_id}}, }, optional_params={}, @@ -650,4 +648,4 @@ def test_update_response_metadata_prices_per_second_deployment_from_its_stamped_ ) assert result._response_ms == pytest.approx(2000) - assert result._hidden_params["response_cost"] == pytest.approx((0.02 + 0.04) * 2) + assert result._hidden_params["response_cost"] == pytest.approx(0.02 * 2) diff --git a/tests/unit/litellm_core_utils/test_get_litellm_params.py b/tests/unit/litellm_core_utils/test_get_litellm_params.py index 9b5771092ac..19a3323ce53 100644 --- a/tests/unit/litellm_core_utils/test_get_litellm_params.py +++ b/tests/unit/litellm_core_utils/test_get_litellm_params.py @@ -21,7 +21,13 @@ from litellm.litellm_core_utils.get_litellm_params import ( from litellm.types.litellm_params import ControlOptions NAMED_PRICE_PARAMS: Final = frozenset( - {"input_cost_per_token", "output_cost_per_token", "input_cost_per_second", "output_cost_per_second"} + { + "input_cost_per_token", + "output_cost_per_token", + "cost_per_second", + "input_cost_per_second", + "output_cost_per_second", + } ) diff --git a/tests/unit/litellm_core_utils/test_litellm_logging.py b/tests/unit/litellm_core_utils/test_litellm_logging.py index 2fc747e1b48..ed614b93a77 100644 --- a/tests/unit/litellm_core_utils/test_litellm_logging.py +++ b/tests/unit/litellm_core_utils/test_litellm_logging.py @@ -495,7 +495,7 @@ class TestZeroCostDiagnostic: DEPLOYMENT_ID: Final = "lit7898-query-only-priced-deployment" MODEL_GROUP: Final = "query-only-priced-chat" QUERY_ONLY_PRICING: Final = {"input_cost_per_query": 0.00042} - PER_SECOND_PRICING: Final = {"input_cost_per_second": 0.00042, "output_cost_per_second": 0.00042} + PER_SECOND_PRICING: Final = {"cost_per_second": 0.00042} FREE_PRICING: Final = {"input_cost_per_token": 0, "output_cost_per_token": 0} @pytest.fixture(params=["query_only", "free"]) @@ -845,7 +845,7 @@ class TestZeroCostDiagnostic: response: Final = self._response(usage) response._response_ms = 1000.0 with caplog.at_level(logging.WARNING, logger="LiteLLM"): - assert logging_obj._response_cost_calculator(result=response) == pytest.approx(0.00084) + assert logging_obj._response_cost_calculator(result=response) == pytest.approx(0.00042) assert logging_obj.model_call_details["zero_cost_diagnostic"] is None assert self._zero_cost_warnings(caplog) == [] diff --git a/tests/unit/router_strategy/test_complexity_router.py b/tests/unit/router_strategy/test_complexity_router.py index 2d66524326f..333524ffffc 100644 --- a/tests/unit/router_strategy/test_complexity_router.py +++ b/tests/unit/router_strategy/test_complexity_router.py @@ -5043,6 +5043,7 @@ class TestRouterPreRoutingAliasOverrides: "model": "auto_router/complexity_router", "input_cost_per_token": 0.0, "output_cost_per_token": 0.0, + "cost_per_second": 0.0, "input_cost_per_second": 0.0, "drop_params": True, "complexity_router_config": {"tiers": {"SIMPLE": "gpt-4o-mini"}}, @@ -5064,7 +5065,12 @@ class TestRouterPreRoutingAliasOverrides: assert result is not None # Non-pricing alias params still carry over. assert request_kwargs["drop_params"] is True - for field in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second"): + for field in ( + "input_cost_per_token", + "output_cost_per_token", + "cost_per_second", + "input_cost_per_second", + ): assert field not in request_kwargs @pytest.mark.asyncio diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index adfedf61d45..c1939a23e74 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -3037,9 +3037,9 @@ def test_completion_cost_logs_cache_and_reasoning_breakdown_for_custom_pricing() @pytest.mark.parametrize("custom_llm_provider", ["together_ai", "openai", "anthropic", "bedrock", "azure"]) def test_cost_per_token_per_second_pricing(monkeypatch, custom_llm_provider: str): """ - Models priced by duration (input/output_cost_per_second) with no per-token rates + Models priced by input/output duration rates with no per-token rates must be billed as cost_per_second * response_time_ms / 1000 in cost_per_token, - whether or not the provider has its own cost calculator. + using only the input rate even when both are set, whether or not the provider has its own calculator. """ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) @@ -3064,11 +3064,40 @@ def test_cost_per_token_per_second_pricing(monkeypatch, custom_llm_provider: str response_time_ms=1500.0, ) - assert prompt_cost == pytest.approx(0.02 * 1.5) - assert completion_cost_value == pytest.approx(0.04 * 1.5) + assert (prompt_cost, completion_cost_value) == pytest.approx((0.02 * 1.5, 0.0)) -def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(monkeypatch): +def test_azure_chat_uses_token_rates_when_output_cost_per_second_is_set( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + model: Final = "test-azure-chat-token-and-output-second-pricing" + litellm.register_model( + model_cost={ + model: { + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "output_cost_per_second": 0.4, + "litellm_provider": "azure", + "mode": "chat", + } + } + ) + + cost: Final = cost_per_token( + model=model, + custom_llm_provider="azure", + prompt_tokens=10, + completion_tokens=20, + response_time_ms=1500.0, + ) + + assert cost == pytest.approx((10 * 1e-6, 20 * 2e-6)) + + +def test_cost_per_token_ignores_cost_per_second_when_token_pricing_is_set(monkeypatch): monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) @@ -3078,8 +3107,7 @@ def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(m model: { "input_cost_per_token": 1e-6, "output_cost_per_token": 2e-6, - "input_cost_per_second": 0.02, - "output_cost_per_second": 0.04, + "cost_per_second": 0.02, "litellm_provider": "openai", "mode": "chat", } @@ -3098,6 +3126,39 @@ def test_cost_per_token_keeps_token_pricing_when_per_second_rates_are_also_set(m assert completion_cost_value == pytest.approx(20 * 2e-6) +@pytest.mark.parametrize( + ("pricing_fields", "expected_rate"), + [ + ({"cost_per_second": 0.02}, 0.02), + ({"output_cost_per_second": 0.04}, 0.04), + ( + {"cost_per_second": 0.05, "input_cost_per_second": 0.02, "output_cost_per_second": 0.04}, + 0.05, + ), + ({"input_cost_per_second": 0.02}, 0.02), + ], +) +def test_cost_per_token_resolves_per_second_rate_precedence( + monkeypatch, pricing_fields: dict[str, float], expected_rate: float +): + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + model: Final = "test-chat-per-second-rate-precedence" + entry: Final = {**pricing_fields, "litellm_provider": "together_ai", "mode": "chat"} + litellm.register_model( + model_cost={model: entry} + ) + + assert cost_per_token( + model=model, + custom_llm_provider="together_ai", + prompt_tokens=10, + completion_tokens=20, + response_time_ms=1500.0, + ) == pytest.approx((expected_rate * 1.5, 0.0)) + + def _logging_obj_with_call_window(duration_ms: float) -> Logging: start_time: Final = datetime.datetime(2026, 9, 21, 12, 0, 0) logging_obj: Final = Logging( @@ -3160,7 +3221,7 @@ def test_completion_cost_per_second_deployment_bills_the_call_duration( litellm_logging_obj=_logging_obj_with_call_window(logged_duration_ms), ) - assert cost == pytest.approx((0.02 + 0.04) * expected_seconds) + assert cost == pytest.approx(0.02 * expected_seconds) @pytest.mark.parametrize("mode", ["audio_transcription", "audio_speech", "video_generation", "realtime"]) diff --git a/tests/unit/test_register_model_custom_pricing.py b/tests/unit/test_register_model_custom_pricing.py index 452a15334ef..fa4fee8f6d8 100644 --- a/tests/unit/test_register_model_custom_pricing.py +++ b/tests/unit/test_register_model_custom_pricing.py @@ -11,6 +11,7 @@ calculations for DB-sourced models with prompt caching pricing. import copy import os +from typing import Final import pytest @@ -993,3 +994,21 @@ def test_completion_cost_applies_off_peak_only_deployment_pricing(): finally: _restore_model_cost_entries(original_entries) del router + + +def test_completion_registers_cost_per_second_pricing(): + model_key: Final = "openai/test-cost-per-second-registration" + original_entries: Final = _snapshot_model_cost_entries([model_key]) + + try: + litellm.completion( + model=model_key, + messages=[{"role": "user", "content": "hello"}], + api_key="fake-key", + cost_per_second=0.02, + mock_response="hello back", + ) + + assert litellm.model_cost[model_key]["cost_per_second"] == 0.02 + finally: + _restore_model_cost_entries(original_entries) diff --git a/tests/unit/test_utils.py b/tests/unit/test_utils.py index 2c612aa350c..ab7ae5ab3b8 100644 --- a/tests/unit/test_utils.py +++ b/tests/unit/test_utils.py @@ -648,6 +648,7 @@ def validate_model_cost_values(model_data, exceptions=None): "output_cost_per_image_4K", "input_cost_per_pixel", "output_cost_per_pixel", + "cost_per_second", "input_cost_per_second", "output_cost_per_second", "output_cost_per_second_480p", @@ -829,6 +830,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "input_cost_per_pixel": {"type": "number"}, "input_cost_per_query": {"type": "number"}, "input_cost_per_request": {"type": "number"}, + "cost_per_second": {"type": "number"}, "input_cost_per_second": {"type": "number"}, "input_cost_per_token": {"type": "number"}, "input_cost_per_token_above_128k_tokens": {"type": "number"}, diff --git a/tests/unit/types/test_router.py b/tests/unit/types/test_router.py index 4d4c326d1ca..4881b094cd6 100644 --- a/tests/unit/types/test_router.py +++ b/tests/unit/types/test_router.py @@ -40,6 +40,7 @@ def test_custom_pricing_params_keeps_every_field_it_had(): "output_cost_per_character", "cache_read_input_token_cost", "cache_creation_input_token_cost", + "cost_per_second", "input_cost_per_second", "cache_read_input_token_cost_flex", "input_cost_per_character_above_128k_tokens", diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 077e6a6ecee..dc0365a14b1 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -33013,6 +33013,8 @@ export interface components { complexity_router_default_model?: string | null; /** Configurable Clientside Auth Params */ configurable_clientside_auth_params?: (string | components["schemas"]["ConfigurableClientsideParamsCustomAuth-Input"])[] | null; + /** Cost Per Second */ + cost_per_second?: number | null; /** Custom Llm Provider */ custom_llm_provider?: string | null; /** Default Api Key Rpm Limit */ @@ -46842,6 +46844,8 @@ export interface components { complexity_router_default_model?: string | null; /** Configurable Clientside Auth Params */ configurable_clientside_auth_params?: (string | components["schemas"]["ConfigurableClientsideParamsCustomAuth-Input"])[] | null; + /** Cost Per Second */ + cost_per_second?: number | null; /** Custom Llm Provider */ custom_llm_provider?: string | null; /** Default Api Key Rpm Limit */