Removing get_model_info from Lemonade provider. Implemented get_models which gets hooked into get_valid_models litellm utility. Also, added a simple cost calculator implementation for Lemonade so calling cost_calculator.completion_cost() doesn't return an error when a model is not found in the model_cost json.

This commit is contained in:
Eddie Richter 2025-09-30 09:29:13 -06:00
parent 1e1e4c36ac
commit 19e7070b73
4 changed files with 73 additions and 34 deletions

View file

@ -58,6 +58,9 @@ from litellm.llms.vertex_ai.cost_calculator import (
)
from litellm.llms.vertex_ai.cost_calculator import cost_router as google_cost_router
from litellm.llms.xai.cost_calculator import cost_per_token as xai_cost_per_token
from litellm.llms.lemonade.cost_calculator import (
cost_per_token as lemonade_cost_per_token,
)
from litellm.responses.utils import ResponseAPILoggingUtils
from litellm.types.llms.openai import (
HttpxBinaryResponseContent,
@ -347,6 +350,8 @@ def cost_per_token( # noqa: PLR0915
return perplexity_cost_per_token(model=model, usage=usage_block)
elif custom_llm_provider == "xai":
return xai_cost_per_token(model=model, usage=usage_block)
elif custom_llm_provider == "lemonade":
return lemonade_cost_per_token(model=model, usage=usage_block)
elif custom_llm_provider == "dashscope":
from litellm.llms.dashscope.cost_calculator import (
cost_per_token as dashscope_cost_per_token,

View file

@ -60,46 +60,45 @@ class LemonadeChatConfig(OpenAILikeChatConfig):
def get_config(cls):
return super().get_config()
def get_model_info(self, model: str) -> ModelInfoBase:
if model.startswith("lemonade/"):
model = model.split("/", 1)[1]
api_base = get_secret_str("LEMONADE_API_BASE") or "http://localhost:8000"
def get_models(self, api_key: Optional[str] = None, api_base: Optional[str] = None):
"""
Get available models from Lemonade API.
This method queries the Lemonade /models endpoint to retrieve the list of available models.
Args:
api_key: Optional API key (Lemonade doesn't require authentication)
api_base: Optional API base URL (defaults to LEMONADE_API_BASE env var or http://localhost:8000)
Returns:
List of model names prefixed with "lemonade/"
"""
api_base, api_key = self._get_openai_compatible_provider_info(
api_base=api_base, api_key=api_key
)
if api_base is None:
raise ValueError(
"LEMONADE_API_BASE is not set. Please set the environment variable to query Lemonade's /models endpoint."
)
# Getting the list of models from lemonade to verify the model exists
# Getting the list of models from lemonade
try:
response = litellm.module_level_client.get(
url=f"{api_base}/api/v1/models",
url=f"{api_base}/models",
)
except Exception as e:
raise Exception(
f"LemonadeError: Error getting model info for {model}. Set Lemonade API Base via `LEMONADE_API_BASE` environment variable. Error: {e}"
)
# Making sure the model exists in lemonade
model_found = False
model_list = response.json().get("data", [])
for model_iter in model_list:
if model_iter['id'] == model:
model_found = True
break
if not model_found:
raise ValueError(
f"LemonadeError: Model {model} not found. Available models: {[m['id'] for m in model_list]}"
f"Failed to fetch models from Lemonade. Set Lemonade API Base via `LEMONADE_API_BASE` environment variable. Error: {e}"
)
# Returning the model if it was found in lemonade. Currently there is no mechanism to report
# the models max input or output tokens so leaving those as None
return ModelInfoBase(
key=model,
litellm_provider="lemonade",
mode="chat",
input_cost_per_token=0.0,
output_cost_per_token=0.0,
max_input_tokens=None,
max_output_tokens=None,
max_tokens=None,
)
if response.status_code != 200:
raise ValueError(
f"Failed to fetch models from Lemonade. Status code: {response.status_code}, Response: {response.text}"
)
model_list = response.json().get("data", [])
return ["lemonade/" + model["id"] for model in model_list]
def _get_openai_compatible_provider_info(
self, api_base: Optional[str], api_key: Optional[str]

View file

@ -0,0 +1,35 @@
"""
Cost calculation for Lemonade LLM provider.
Since Lemonade is a local/self-hosted service, all costs default to 0.
This prevents cost calculation errors when using models not in model_prices_and_context_window.json
"""
from typing import Tuple
from litellm.types.utils import Usage
def cost_per_token(
model: str,
usage: Usage,
) -> Tuple[float, float]:
"""
Calculate cost per token for Lemonade models.
Since Lemonade is a local/self-hosted deployment, there are no per-token costs.
This function returns (0.0, 0.0) for all models to allow cost tracking to work
without errors for any Lemonade model, regardless of whether it's in the
model_prices_and_context_window.json file.
Args:
model: The model name (with or without "lemonade/" prefix)
usage: Usage object containing token counts
Returns:
Tuple of (prompt_cost, completion_cost) - always (0.0, 0.0) for Lemonade
"""
# Lemonade is self-hosted/local, so cost is always 0
prompt_cost = 0.0
completion_cost = 0.0
return prompt_cost, completion_cost

View file

@ -4775,8 +4775,6 @@ def _get_model_info_helper( # noqa: PLR0915
custom_llm_provider == "ollama" or custom_llm_provider == "ollama_chat"
) and not _is_potential_model_name_in_model_cost(potential_model_names):
return litellm.OllamaConfig().get_model_info(model)
elif (custom_llm_provider == "lemonade" and not _is_potential_model_name_in_model_cost(potential_model_names)):
return litellm.LemonadeChatConfig().get_model_info(model)
else:
"""
Check if: (in order of specificity)
@ -7319,6 +7317,8 @@ class ProviderConfigManager:
)
return VLLMModelInfo()
elif LlmProviders.LEMONADE == provider:
return litellm.LemonadeChatConfig()
return None
@staticmethod