From 929510ef5d9d329a6993f7cb4afa79903876ffa2 Mon Sep 17 00:00:00 2001 From: Eddie Richter Date: Tue, 23 Sep 2025 21:13:52 -0600 Subject: [PATCH] Adding unit tests and documentation --- docs/my-website/docs/providers/lemonade.md | 188 ++++++++++++++++++ docs/my-website/sidebars.js | 1 + .../llms/lemonade/test_lemonade.py | 180 +++++++++++++++++ 3 files changed, 369 insertions(+) create mode 100644 docs/my-website/docs/providers/lemonade.md create mode 100644 tests/test_litellm/llms/lemonade/test_lemonade.py diff --git a/docs/my-website/docs/providers/lemonade.md b/docs/my-website/docs/providers/lemonade.md new file mode 100644 index 00000000000..87d41902a07 --- /dev/null +++ b/docs/my-website/docs/providers/lemonade.md @@ -0,0 +1,188 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Lemonade + +Lemonade is an OpenAI-compatible AI provider that offers local language model inference on AMD Ryzen AI models. The provider supports standard chat completions with full OpenAI API compatibility. + +| Property | Details | +|-------|-------| +| Description | OpenAI-compatible AI provider for local and cloud-based language model inference | +| Provider Route on LiteLLM | `lemonade/` (add this prefix to the model name - e.g. `lemonade/your-model-name`) | +| API Endpoint for Provider | http://localhost:8000/api/v1 (default) | +| Supported Endpoints | `/chat/completions` | + +## Supported OpenAI Parameters + +Lemonade is fully OpenAI-compatible and supports the following parameters: + +``` +"repeat_penalty" +"functions" +"logit_bias" +"max_tokens" +"max_completion_tokens" +"presence_penalty" +"stop" +"temperature" +"top_p" +"top_k" +"response_format" +"tools" +``` + + +## API Key Setup + +Lemonade can be configured with custom API URLs and doesn't require strict API key validation. Set the `LEMONADE_API_BASE` environment variable to modify the base URL. + +## Usage + + + + +```python +from litellm import completion +import os + +# Optional: Set custom API base. Useful if your lemonade server is on +# a different port +os.environ['LEMONADE_API_BASE'] = "http://localhost:8000/api/v1" + +response = completion( + model="lemonade/your-model-name", + messages=[ + {"role": "user", "content": "Hello from LiteLLM!"} + ], +) +print(response) +``` + +## Streaming + +```python +from litellm import completion +import os + +# Optional: Set custom API base. Useful if your lemonade server is on +# a different port +os.environ['LEMONADE_API_BASE'] = "http://localhost:8000/api/v1" + +response = completion( + model="lemonade/your-model-name", + messages=[ + {"role": "user", "content": "Write a short story"} + ], + stream=True +) + +for chunk in response: + print(chunk.choices[0].delta.content, end='', flush=True) +``` + +## Advanced Usage + +### Custom Parameters + +Lemonade supports additional parameters beyond the standard OpenAI set: + +```python +from litellm import completion + +response = completion( + model="lemonade/your-model-name", + messages=[{"role": "user", "content": "Explain quantum computing"}], + temperature=0.7, + max_tokens=500, + top_p=0.9, + top_k=50, + repeat_penalty=1.1, + stop=["Human:", "AI:"] +) +print(response) +``` + +### Function Calling + +Lemonade supports OpenAI-compatible function calling: + +```python +from litellm import completion + +functions = [ + { + "name": "get_weather", + "description": "Get current weather information", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state" + } + }, + "required": ["location"] + } + } +] + +response = completion( + model="lemonade/your-model-name", + messages=[{"role": "user", "content": "What's the weather in San Francisco?"}], + tools=[{"type": "function", "function": f} for f in functions], + tool_choice="auto" +) +print(response) +``` + +### Response Format + +Lemonade supports structured output with response format: + +```python +from litellm import completion +import json + +# Define schema in response_format +response = completion( + model="lemonade/Qwen3-Coder-30B-A3B-Instruct-GGUF", + messages=[{"role": "user", "content": "Generate JSON data for a person with their name, age, and city."}], + response_format={ + "type": "json_schema", + "json_schema": { + "name": "person", + "schema": { + "type": "object", + "properties": { + "name": {"type": "string"}, + "age": {"type": "integer"}, + "city": {"type": "string"} + }, + "required": ["name", "age"] + } + } + } +) + +print(f"Model: {response.model}") +print(f"JSON Output:") +json_data = json.loads(response.choices[0].message.content) +print(json.dumps(json_data, indent=2)) +``` + +## Available Models + +Lemonade automatically validates available models by querying the `/models` endpoint. You can check available models programmatically: + +```python +import httpx + +api_base = "http://localhost:8000" # or your custom base +response = httpx.get(f"{api_base}/api/v1/models") +models = response.json() +print("Available models:", [model['id'] for model in models.get('data', [])]) +``` + +## Support + +For more information regarding Lemonade please go to to the [Lemonade website](https://lemonade-server.ai/) or [Lemonade repository](https://github.com/lemonade-sdk/lemonade). diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 9ca0201ff71..d450159f934 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -477,6 +477,7 @@ const sidebars = { "providers/fireworks_ai", "providers/clarifai", "providers/compactifai", + "providers/lemonade", "providers/vllm", "providers/llamafile", "providers/infinity", diff --git a/tests/test_litellm/llms/lemonade/test_lemonade.py b/tests/test_litellm/llms/lemonade/test_lemonade.py new file mode 100644 index 00000000000..d3850d6271d --- /dev/null +++ b/tests/test_litellm/llms/lemonade/test_lemonade.py @@ -0,0 +1,180 @@ +import json +import os +import sys + +import pytest + +sys.path.insert( + 0, os.path.abspath("../../../../..") +) # Adds the parent directory to the system path +from unittest.mock import MagicMock, patch + +from litellm.llms.lemonade.chat.transformation import LemonadeChatConfig +from litellm.types.utils import ModelResponse +import httpx + + +def test_lemonade_config_initialization(): + """Test that LemonadeChatConfig can be initialized with various parameters""" + config = LemonadeChatConfig( + temperature=0.7, + max_tokens=100, + top_p=0.9, + top_k=50, + repeat_penalty=1.1 + ) + + assert config.custom_llm_provider == "lemonade" + assert config.temperature == 0.7 + assert config.max_tokens == 100 + assert config.top_p == 0.9 + assert config.top_k == 50 + assert config.repeat_penalty == 1.1 + + +def test_get_openai_compatible_provider_info(): + """Test the provider info method returns correct API base and key""" + config = LemonadeChatConfig() + + api_base, key = config._get_openai_compatible_provider_info( + api_base=None, + api_key=None + ) + + assert api_base == "http://localhost:8000/api/v1" + assert key == "lemonade" + + +def test_get_openai_compatible_provider_info_with_custom_base(): + """Test the provider info method with custom API base""" + config = LemonadeChatConfig() + + custom_api_base = "https://custom.lemonade.ai/v1" + api_base, key = config._get_openai_compatible_provider_info( + api_base=custom_api_base, + api_key=None + ) + + assert api_base == custom_api_base + assert key == "lemonade" + + +def test_transform_response(): + """Test the response transformation adds lemonade prefix to model name""" + config = LemonadeChatConfig() + + # Mock raw response + raw_response = MagicMock() + raw_response.status_code = 200 + raw_response.headers = {} + + # Create a model response + model_response = ModelResponse() + + # Mock the parent class transform_response method + with patch.object(config.__class__.__bases__[0], 'transform_response') as mock_parent: + mock_parent.return_value = model_response + + result = config.transform_response( + model="test-model", + raw_response=raw_response, + model_response=model_response, + logging_obj=MagicMock(), + request_data={}, + messages=[], + optional_params={}, + litellm_params={}, + encoding=None, + api_key="test-key", + json_mode=False, + ) + + # Check that the model name is prefixed with "lemonade/" + assert hasattr(result, 'model') + assert result.model == "lemonade/test-model" + + +def test_config_get_config(): + """Test that get_config method returns the configuration""" + config_dict = LemonadeChatConfig.get_config() + assert isinstance(config_dict, dict) + + +def test_response_format_support(): + """Test that response_format parameter is supported""" + response_format = { + "type": "json_object" + } + + config = LemonadeChatConfig(response_format=response_format) + assert config.response_format == response_format + + +def test_tools_support(): + """Test that tools parameter is supported""" + tools = [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get weather information" + } + } + ] + + config = LemonadeChatConfig(tools=tools) + assert config.tools == tools + + +def test_functions_support(): + """Test that functions parameter is supported""" + functions = [ + { + "name": "get_weather", + "description": "Get weather information", + "parameters": { + "type": "object", + "properties": {} + } + } + ] + + config = LemonadeChatConfig(functions=functions) + assert config.functions == functions + + +def test_stop_parameter_support(): + """Test that stop parameter supports both string and list""" + # Test with string + config1 = LemonadeChatConfig(stop="STOP") + assert config1.stop == "STOP" + + # Test with list + config2 = LemonadeChatConfig(stop=["STOP", "END"]) + assert config2.stop == ["STOP", "END"] + + +def test_logit_bias_support(): + """Test that logit_bias parameter is supported""" + logit_bias = {"50256": -100} + + config = LemonadeChatConfig(logit_bias=logit_bias) + assert config.logit_bias == logit_bias + + +def test_presence_penalty_support(): + """Test that presence_penalty parameter is supported""" + config = LemonadeChatConfig(presence_penalty=0.5) + assert config.presence_penalty == 0.5 + + +def test_n_parameter_support(): + """Test that n parameter (number of completions) is supported""" + config = LemonadeChatConfig(n=3) + assert config.n == 3 + + +def test_max_completion_tokens_support(): + """Test that max_completion_tokens parameter is supported""" + config = LemonadeChatConfig(max_completion_tokens=150) + assert config.max_completion_tokens == 150 \ No newline at end of file