From 19e7070b737a25cca3cc9eed0a382b42efee0d8e Mon Sep 17 00:00:00 2001 From: Eddie Richter Date: Tue, 30 Sep 2025 09:29:13 -0600 Subject: [PATCH] Removing get_model_info from Lemonade provider. Implemented get_models which gets hooked into get_valid_models litellm utility. Also, added a simple cost calculator implementation for Lemonade so calling cost_calculator.completion_cost() doesn't return an error when a model is not found in the model_cost json. --- litellm/cost_calculator.py | 5 ++ litellm/llms/lemonade/chat/transformation.py | 63 ++++++++++---------- litellm/llms/lemonade/cost_calculator.py | 35 +++++++++++ litellm/utils.py | 4 +- 4 files changed, 73 insertions(+), 34 deletions(-) create mode 100644 litellm/llms/lemonade/cost_calculator.py diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 94a9523facd..4bb14eb8391 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -58,6 +58,9 @@ from litellm.llms.vertex_ai.cost_calculator import ( ) from litellm.llms.vertex_ai.cost_calculator import cost_router as google_cost_router from litellm.llms.xai.cost_calculator import cost_per_token as xai_cost_per_token +from litellm.llms.lemonade.cost_calculator import ( + cost_per_token as lemonade_cost_per_token, +) from litellm.responses.utils import ResponseAPILoggingUtils from litellm.types.llms.openai import ( HttpxBinaryResponseContent, @@ -347,6 +350,8 @@ def cost_per_token( # noqa: PLR0915 return perplexity_cost_per_token(model=model, usage=usage_block) elif custom_llm_provider == "xai": return xai_cost_per_token(model=model, usage=usage_block) + elif custom_llm_provider == "lemonade": + return lemonade_cost_per_token(model=model, usage=usage_block) elif custom_llm_provider == "dashscope": from litellm.llms.dashscope.cost_calculator import ( cost_per_token as dashscope_cost_per_token, diff --git a/litellm/llms/lemonade/chat/transformation.py b/litellm/llms/lemonade/chat/transformation.py index f1dc0e1a37d..094e625fe64 100644 --- a/litellm/llms/lemonade/chat/transformation.py +++ b/litellm/llms/lemonade/chat/transformation.py @@ -60,46 +60,45 @@ class LemonadeChatConfig(OpenAILikeChatConfig): def get_config(cls): return super().get_config() - def get_model_info(self, model: str) -> ModelInfoBase: - if model.startswith("lemonade/"): - model = model.split("/", 1)[1] - api_base = get_secret_str("LEMONADE_API_BASE") or "http://localhost:8000" + def get_models(self, api_key: Optional[str] = None, api_base: Optional[str] = None): + """ + Get available models from Lemonade API. + + This method queries the Lemonade /models endpoint to retrieve the list of available models. + + Args: + api_key: Optional API key (Lemonade doesn't require authentication) + api_base: Optional API base URL (defaults to LEMONADE_API_BASE env var or http://localhost:8000) + + Returns: + List of model names prefixed with "lemonade/" + """ + api_base, api_key = self._get_openai_compatible_provider_info( + api_base=api_base, api_key=api_key + ) + + if api_base is None: + raise ValueError( + "LEMONADE_API_BASE is not set. Please set the environment variable to query Lemonade's /models endpoint." + ) - # Getting the list of models from lemonade to verify the model exists + # Getting the list of models from lemonade try: response = litellm.module_level_client.get( - url=f"{api_base}/api/v1/models", + url=f"{api_base}/models", ) except Exception as e: - raise Exception( - f"LemonadeError: Error getting model info for {model}. Set Lemonade API Base via `LEMONADE_API_BASE` environment variable. Error: {e}" - ) - - # Making sure the model exists in lemonade - model_found = False - model_list = response.json().get("data", []) - for model_iter in model_list: - if model_iter['id'] == model: - model_found = True - break - - if not model_found: raise ValueError( - f"LemonadeError: Model {model} not found. Available models: {[m['id'] for m in model_list]}" + f"Failed to fetch models from Lemonade. Set Lemonade API Base via `LEMONADE_API_BASE` environment variable. Error: {e}" ) - # Returning the model if it was found in lemonade. Currently there is no mechanism to report - # the models max input or output tokens so leaving those as None - return ModelInfoBase( - key=model, - litellm_provider="lemonade", - mode="chat", - input_cost_per_token=0.0, - output_cost_per_token=0.0, - max_input_tokens=None, - max_output_tokens=None, - max_tokens=None, - ) + if response.status_code != 200: + raise ValueError( + f"Failed to fetch models from Lemonade. Status code: {response.status_code}, Response: {response.text}" + ) + + model_list = response.json().get("data", []) + return ["lemonade/" + model["id"] for model in model_list] def _get_openai_compatible_provider_info( self, api_base: Optional[str], api_key: Optional[str] diff --git a/litellm/llms/lemonade/cost_calculator.py b/litellm/llms/lemonade/cost_calculator.py new file mode 100644 index 00000000000..27e1ca275f8 --- /dev/null +++ b/litellm/llms/lemonade/cost_calculator.py @@ -0,0 +1,35 @@ +""" +Cost calculation for Lemonade LLM provider. + +Since Lemonade is a local/self-hosted service, all costs default to 0. +This prevents cost calculation errors when using models not in model_prices_and_context_window.json +""" +from typing import Tuple + +from litellm.types.utils import Usage + + +def cost_per_token( + model: str, + usage: Usage, +) -> Tuple[float, float]: + """ + Calculate cost per token for Lemonade models. + + Since Lemonade is a local/self-hosted deployment, there are no per-token costs. + This function returns (0.0, 0.0) for all models to allow cost tracking to work + without errors for any Lemonade model, regardless of whether it's in the + model_prices_and_context_window.json file. + + Args: + model: The model name (with or without "lemonade/" prefix) + usage: Usage object containing token counts + + Returns: + Tuple of (prompt_cost, completion_cost) - always (0.0, 0.0) for Lemonade + """ + # Lemonade is self-hosted/local, so cost is always 0 + prompt_cost = 0.0 + completion_cost = 0.0 + + return prompt_cost, completion_cost diff --git a/litellm/utils.py b/litellm/utils.py index fbc5d09e57b..8dfa2416a62 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4775,8 +4775,6 @@ def _get_model_info_helper( # noqa: PLR0915 custom_llm_provider == "ollama" or custom_llm_provider == "ollama_chat" ) and not _is_potential_model_name_in_model_cost(potential_model_names): return litellm.OllamaConfig().get_model_info(model) - elif (custom_llm_provider == "lemonade" and not _is_potential_model_name_in_model_cost(potential_model_names)): - return litellm.LemonadeChatConfig().get_model_info(model) else: """ Check if: (in order of specificity) @@ -7319,6 +7317,8 @@ class ProviderConfigManager: ) return VLLMModelInfo() + elif LlmProviders.LEMONADE == provider: + return litellm.LemonadeChatConfig() return None @staticmethod