From 407639cc7d2a5a54162ef0f1c4bb17c49158fbb6 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 5 Jul 2024 20:58:08 -0700 Subject: [PATCH 1/4] fix(cost_calculator.py): support openai+azure tts calls --- litellm/cost_calculator.py | 48 ++++++++++- .../litellm_core_utils/llm_cost_calc/utils.py | 85 +++++++++++++++++++ ...odel_prices_and_context_window_backup.json | 22 ++++- litellm/tests/test_completion_cost.py | 15 +++- litellm/utils.py | 4 +- model_prices_and_context_window.json | 22 ++++- 6 files changed, 191 insertions(+), 5 deletions(-) create mode 100644 litellm/litellm_core_utils/llm_cost_calc/utils.py diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 062e98be97b..e4963a6f1a3 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -13,6 +13,7 @@ from litellm.litellm_core_utils.llm_cost_calc.google import ( from litellm.litellm_core_utils.llm_cost_calc.google import ( cost_per_token as google_cost_per_token, ) +from litellm.litellm_core_utils.llm_cost_calc.utils import _generic_cost_per_character from litellm.utils import ( CallTypes, CostPerToken, @@ -62,6 +63,23 @@ def cost_per_token( ### CUSTOM PRICING ### custom_cost_per_token: Optional[CostPerToken] = None, custom_cost_per_second: Optional[float] = None, + ### CALL TYPE ### + call_type: Literal[ + "embedding", + "aembedding", + "completion", + "acompletion", + "atext_completion", + "text_completion", + "image_generation", + "aimage_generation", + "moderation", + "amoderation", + "atranscription", + "transcription", + "aspeech", + "speech", + ] = "completion", ) -> Tuple[float, float]: """ Calculates the cost per token for a given model, prompt tokens, and completion tokens. @@ -76,6 +94,7 @@ def cost_per_token( custom_llm_provider (str): The llm provider to whom the call was made (see init.py for full list) custom_cost_per_token: Optional[CostPerToken]: the cost per input + output token for the llm api call. custom_cost_per_second: Optional[float]: the cost per second for the llm api call. + call_type: Optional[str]: the call type Returns: tuple: A tuple containing the cost in USD dollars for prompt tokens and completion tokens, respectively. @@ -159,6 +178,27 @@ def cost_per_token( prompt_tokens=prompt_tokens, completion_tokens=completion_tokens, ) + elif call_type == "speech" or call_type == "aspeech": + prompt_cost, completion_cost = _generic_cost_per_character( + model=model_without_prefix, + custom_llm_provider=custom_llm_provider, + prompt_characters=prompt_characters, + completion_characters=completion_characters, + custom_prompt_cost=None, + custom_completion_cost=0, + ) + if prompt_cost is None or completion_cost is None: + raise ValueError( + "cost for tts call is None. prompt_cost={}, completion_cost={}, model={}, custom_llm_provider={}, prompt_characters={}, completion_characters={}".format( + prompt_cost, + completion_cost, + model_without_prefix, + custom_llm_provider, + prompt_characters, + completion_characters, + ) + ) + return prompt_cost, completion_cost elif model in model_cost_ref: print_verbose(f"Success: model={model} in model_cost_map") print_verbose( @@ -289,7 +329,7 @@ def cost_per_token( return prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar else: # if model is not in model_prices_and_context_window.json. Raise an exception-let users know - error_str = f"Model not in model_prices_and_context_window.json. You passed model={model}. Register pricing for model - https://docs.litellm.ai/docs/proxy/custom_pricing\n" + error_str = f"Model not in model_prices_and_context_window.json. You passed model={model}, custom_llm_provider={custom_llm_provider}. Register pricing for model - https://docs.litellm.ai/docs/proxy/custom_pricing\n" raise litellm.exceptions.NotFoundError( # type: ignore message=error_str, model=model, @@ -535,6 +575,11 @@ def completion_cost( raise Exception( f"Model={image_gen_model_name} not found in completion cost model map" ) + elif ( + call_type == CallTypes.speech.value or call_type == CallTypes.aspeech.value + ): + prompt_characters = litellm.utils._count_characters(text=prompt) + # Calculate cost based on prompt_tokens, completion_tokens if ( "togethercomputer" in model @@ -591,6 +636,7 @@ def completion_cost( custom_cost_per_token=custom_cost_per_token, prompt_characters=prompt_characters, completion_characters=completion_characters, + call_type=call_type, ) _final_cost = prompt_tokens_cost_usd_dollar + completion_tokens_cost_usd_dollar print_verbose( diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py new file mode 100644 index 00000000000..e986a22a6c9 --- /dev/null +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -0,0 +1,85 @@ +# What is this? +## Helper utilities for cost_per_token() + +import traceback +from typing import List, Literal, Optional, Tuple + +import litellm +from litellm import verbose_logger + + +def _generic_cost_per_character( + model: str, + custom_llm_provider: str, + prompt_characters: float, + completion_characters: float, + custom_prompt_cost: Optional[float], + custom_completion_cost: Optional[float], +) -> Tuple[Optional[float], Optional[float]]: + """ + Generic function to help calculate cost per character. + """ + """ + Calculates the cost per character for a given model, input messages, and response object. + + Input: + - model: str, the model name without provider prefix + - custom_llm_provider: str, "vertex_ai-*" + - prompt_characters: float, the number of input characters + - completion_characters: float, the number of output characters + + Returns: + Tuple[Optional[float], Optional[float]] - prompt_cost_in_usd, completion_cost_in_usd. + - returns None if not able to calculate cost. + + Raises: + Exception if 'input_cost_per_character' or 'output_cost_per_character' is missing from model_info + """ + args = locals() + ## GET MODEL INFO + model_info = litellm.get_model_info( + model=model, custom_llm_provider=custom_llm_provider + ) + + ## CALCULATE INPUT COST + try: + if custom_prompt_cost is None: + assert ( + "input_cost_per_character" in model_info + and model_info["input_cost_per_character"] is not None + ), "model info for model={} does not have 'input_cost_per_character'-pricing\nmodel_info={}".format( + model, model_info + ) + custom_prompt_cost = model_info["input_cost_per_character"] + + prompt_cost = prompt_characters * custom_prompt_cost + except Exception as e: + verbose_logger.error( + "litellm.litellm_core_utils.llm_cost_calc.utils.py::cost_per_character(): Exception occured - {}\n{}\nDefaulting to None".format( + str(e), traceback.format_exc() + ) + ) + + prompt_cost = None + + ## CALCULATE OUTPUT COST + try: + if custom_completion_cost is None: + assert ( + "output_cost_per_character" in model_info + and model_info["output_cost_per_character"] is not None + ), "model info for model={} does not have 'output_cost_per_character'-pricing\nmodel_info={}".format( + model, model_info + ) + custom_completion_cost = model_info["output_cost_per_character"] + completion_cost = completion_characters * custom_completion_cost + except Exception as e: + verbose_logger.error( + "litellm.litellm_core_utils.llm_cost_calc.utils.py::cost_per_character(): Exception occured - {}\n{}\nDefaulting to None".format( + str(e), traceback.format_exc() + ) + ) + + completion_cost = None + + return prompt_cost, completion_cost diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 34b1613445e..be2fab51d12 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -397,7 +397,27 @@ "input_cost_per_second": 0, "output_cost_per_second": 0.0001, "litellm_provider": "openai" - }, + }, + "tts-1": { + "mode": "audio_speech", + "input_cost_per_character": 0.000015, + "litellm_provider": "openai" + }, + "tts-1-hd": { + "mode": "audio_speech", + "input_cost_per_character": 0.000030, + "litellm_provider": "openai" + }, + "azure/tts-1": { + "mode": "audio_speech", + "input_cost_per_character": 0.000015, + "litellm_provider": "azure" + }, + "azure/tts-1-hd": { + "mode": "audio_speech", + "input_cost_per_character": 0.000030, + "litellm_provider": "azure" + }, "azure/whisper-1": { "mode": "audio_transcription", "input_cost_per_second": 0, diff --git a/litellm/tests/test_completion_cost.py b/litellm/tests/test_completion_cost.py index bffb68e0e5d..1b4df0ecc08 100644 --- a/litellm/tests/test_completion_cost.py +++ b/litellm/tests/test_completion_cost.py @@ -712,7 +712,6 @@ def test_vertex_ai_claude_completion_cost(): assert cost == predicted_cost - @pytest.mark.parametrize("sync_mode", [True, False]) @pytest.mark.asyncio async def test_completion_cost_hidden_params(sync_mode): @@ -732,6 +731,7 @@ async def test_completion_cost_hidden_params(sync_mode): assert "response_cost" in response._hidden_params assert isinstance(response._hidden_params["response_cost"], float) + def test_vertex_ai_gemini_predict_cost(): model = "gemini-1.5-flash" messages = [{"role": "user", "content": "Hey, hows it going???"}] @@ -739,3 +739,16 @@ def test_vertex_ai_gemini_predict_cost(): assert predictive_cost > 0 + +@pytest.mark.parametrize("model", ["openai/tts-1", "azure/tts-1"]) +def test_completion_cost_tts(model): + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + cost = completion_cost( + model=model, + prompt="the quick brown fox jumped over the lazy dogs", + call_type="speech", + ) + + assert cost > 0 diff --git a/litellm/utils.py b/litellm/utils.py index 490b809a1cf..50e31053da9 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4705,7 +4705,9 @@ def get_model_info(model: str, custom_llm_provider: Optional[str] = None) -> Mod ) except Exception: raise Exception( - "This model isn't mapped yet. Add it here - https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json" + "This model isn't mapped yet. model={}, custom_llm_provider={}. Add it here - https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json".format( + model, custom_llm_provider + ) ) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 34b1613445e..be2fab51d12 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -397,7 +397,27 @@ "input_cost_per_second": 0, "output_cost_per_second": 0.0001, "litellm_provider": "openai" - }, + }, + "tts-1": { + "mode": "audio_speech", + "input_cost_per_character": 0.000015, + "litellm_provider": "openai" + }, + "tts-1-hd": { + "mode": "audio_speech", + "input_cost_per_character": 0.000030, + "litellm_provider": "openai" + }, + "azure/tts-1": { + "mode": "audio_speech", + "input_cost_per_character": 0.000015, + "litellm_provider": "azure" + }, + "azure/tts-1-hd": { + "mode": "audio_speech", + "input_cost_per_character": 0.000030, + "litellm_provider": "azure" + }, "azure/whisper-1": { "mode": "audio_transcription", "input_cost_per_second": 0, From 6e43cdcb176c769c2bef800d70143c1f2a66d462 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 5 Jul 2024 22:09:08 -0700 Subject: [PATCH 2/4] feat(litellm_logging.py): support cost tracking for tts calls --- litellm/cost_calculator.py | 17 ++++-- litellm/litellm_core_utils/litellm_logging.py | 56 ++++++++++++------- litellm/proxy/_new_secret_config.yaml | 11 ++-- litellm/types/router.py | 7 ++- 4 files changed, 58 insertions(+), 33 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index e4963a6f1a3..09f1375ca83 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -4,6 +4,8 @@ import time import traceback from typing import List, Literal, Optional, Tuple, Union +from pydantic import BaseModel + import litellm import litellm._logging from litellm import verbose_logger @@ -14,6 +16,8 @@ from litellm.litellm_core_utils.llm_cost_calc.google import ( cost_per_token as google_cost_per_token, ) from litellm.litellm_core_utils.llm_cost_calc.utils import _generic_cost_per_character +from litellm.types.llms.openai import HttpxBinaryResponseContent +from litellm.types.router import SPECIAL_MODEL_INFO_PARAMS from litellm.utils import ( CallTypes, CostPerToken, @@ -469,7 +473,9 @@ def completion_cost( prompt_characters = 0 completion_tokens = 0 completion_characters = 0 - if completion_response is not None: + if completion_response is not None and isinstance( + completion_response, BaseModel + ): # get input/output tokens from completion_response prompt_tokens = completion_response.get("usage", {}).get("prompt_tokens", 0) completion_tokens = completion_response.get("usage", {}).get( @@ -654,6 +660,7 @@ def response_cost_calculator( ImageResponse, TranscriptionResponse, TextCompletionResponse, + HttpxBinaryResponseContent, ], model: str, custom_llm_provider: Optional[str], @@ -687,7 +694,8 @@ def response_cost_calculator( if cache_hit is not None and cache_hit is True: response_cost = 0.0 else: - response_object._hidden_params["optional_params"] = optional_params + if isinstance(response_object, BaseModel): + response_object._hidden_params["optional_params"] = optional_params if isinstance(response_object, ImageResponse): response_cost = completion_cost( completion_response=response_object, @@ -697,12 +705,11 @@ def response_cost_calculator( ) else: if ( - model in litellm.model_cost - and custom_pricing is not None - and custom_llm_provider is True + model in litellm.model_cost or custom_pricing is True ): # override defaults if custom pricing is set base_model = model # base_model defaults to None if not set on model_info + response_cost = completion_cost( completion_response=response_object, call_type=call_type, diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 4edbce5e15b..4382a1fcb52 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -24,6 +24,8 @@ from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.redact_messages import ( redact_message_input_output_from_logging, ) +from litellm.types.llms.openai import HttpxBinaryResponseContent +from litellm.types.router import SPECIAL_MODEL_INFO_PARAMS from litellm.types.utils import ( CallTypes, EmbeddingResponse, @@ -521,33 +523,36 @@ class Logging: self.model_call_details["cache_hit"] = cache_hit ## if model in model cost map - log the response cost ## else set cost to None - verbose_logger.debug(f"Model={self.model};") if ( - result is not None - and ( + result is not None and self.stream is not True + ): # handle streaming separately + if ( isinstance(result, ModelResponse) or isinstance(result, EmbeddingResponse) or isinstance(result, ImageResponse) or isinstance(result, TranscriptionResponse) or isinstance(result, TextCompletionResponse) - ) - and self.stream != True - ): # handle streaming separately - self.model_call_details["response_cost"] = ( - litellm.response_cost_calculator( - response_object=result, - model=self.model, - cache_hit=self.model_call_details.get("cache_hit", False), - custom_llm_provider=self.model_call_details.get( - "custom_llm_provider", None - ), - base_model=_get_base_model_from_metadata( - model_call_details=self.model_call_details - ), - call_type=self.call_type, - optional_params=self.optional_params, + or isinstance(result, HttpxBinaryResponseContent) # tts + ): + custom_pricing = use_custom_pricing_for_model( + litellm_params=self.litellm_params + ) + self.model_call_details["response_cost"] = ( + litellm.response_cost_calculator( + response_object=result, + model=self.model, + cache_hit=self.model_call_details.get("cache_hit", False), + custom_llm_provider=self.model_call_details.get( + "custom_llm_provider", None + ), + base_model=_get_base_model_from_metadata( + model_call_details=self.model_call_details + ), + call_type=self.call_type, + optional_params=self.optional_params, + custom_pricing=custom_pricing, + ) ) - ) else: # streaming chunks + image gen. self.model_call_details["response_cost"] = None @@ -2003,3 +2008,14 @@ def get_custom_logger_compatible_class( if isinstance(callback, _PROXY_DynamicRateLimitHandler): return callback # type: ignore return None + + +def use_custom_pricing_for_model(litellm_params: dict) -> bool: + model_info: Optional[dict] = litellm_params.get("metadata", {}).get( + "model_info", {} + ) + if model_info is not None: + for k, v in model_info.items(): + if k in SPECIAL_MODEL_INFO_PARAMS: + return True + return False diff --git a/litellm/proxy/_new_secret_config.yaml b/litellm/proxy/_new_secret_config.yaml index 7f4b86ec401..99f2cf16a65 100644 --- a/litellm/proxy/_new_secret_config.yaml +++ b/litellm/proxy/_new_secret_config.yaml @@ -1,12 +1,9 @@ model_list: - - model_name: "*" + - model_name: tts litellm_params: - model: "openai/*" - mock_response: "Hello world!" - -litellm_settings: - success_callback: ["langfuse"] - failure_callback: ["langfuse"] + model: openai/tts-1 + api_key: os.environ/OPENAI_API_KEY + input_cost_per_character: 0.000015, general_settings: alerting: ["slack"] diff --git a/litellm/types/router.py b/litellm/types/router.py index 78d516d6c79..fb2c82c9784 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -324,7 +324,12 @@ class DeploymentTypedDict(TypedDict): litellm_params: LiteLLMParamsTypedDict -SPECIAL_MODEL_INFO_PARAMS = ["input_cost_per_token", "output_cost_per_token"] +SPECIAL_MODEL_INFO_PARAMS = [ + "input_cost_per_token", + "output_cost_per_token", + "input_cost_per_character", + "output_cost_per_character", +] class Deployment(BaseModel): From 1b1420258243523d8b0f16ff4f0f3e5a8474c34c Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Sat, 6 Jul 2024 11:20:36 -0700 Subject: [PATCH 3/4] fix(litellm_logging.py): fix 'use_custom_pricing_for_model' helper function --- litellm/litellm_core_utils/litellm_logging.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 2bac73b98da..f9f32552da0 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -2019,10 +2019,13 @@ def get_custom_logger_compatible_class( return None -def use_custom_pricing_for_model(litellm_params: dict) -> bool: - model_info: Optional[dict] = litellm_params.get("metadata", {}).get( - "model_info", {} - ) +def use_custom_pricing_for_model(litellm_params: Optional[dict]) -> bool: + if litellm_params is None: + return False + metadata: Optional[dict] = litellm_params.get("metadata", {}) + if metadata is None: + return False + model_info: Optional[dict] = metadata.get("model_info", {}) if model_info is not None: for k, v in model_info.items(): if k in SPECIAL_MODEL_INFO_PARAMS: From f62884da14c698902d600a618d452ee875549cbd Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Sat, 6 Jul 2024 12:28:46 -0700 Subject: [PATCH 4/4] fix(cost_calculator.py): fix completion_response check --- litellm/cost_calculator.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 09f1375ca83..5d976483a9e 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -473,9 +473,10 @@ def completion_cost( prompt_characters = 0 completion_tokens = 0 completion_characters = 0 - if completion_response is not None and isinstance( - completion_response, BaseModel - ): + if completion_response is not None and ( + isinstance(completion_response, BaseModel) + or isinstance(completion_response, dict) + ): # tts returns a custom class # get input/output tokens from completion_response prompt_tokens = completion_response.get("usage", {}).get("prompt_tokens", 0) completion_tokens = completion_response.get("usage", {}).get(