From 97894f1603942ab4784855ba9fd741722a65ac5b Mon Sep 17 00:00:00 2001 From: yrk <2493404415@qq.com> Date: Thu, 14 Aug 2025 10:22:17 +0800 Subject: [PATCH] add ModelScope API support --- docs/my-website/docs/providers/modelscope.md | 242 ++++++++++++++++++ litellm/__init__.py | 39 ++- litellm/constants.py | 88 ++++--- .../get_llm_provider_logic.py | 10 + .../llms/modelscope/chat/transformation.py | 75 ++++++ litellm/types/utils.py | 1 + litellm/utils.py | 6 + .../test_modelscope_chat_transformation.py | 114 +++++++++ 8 files changed, 531 insertions(+), 44 deletions(-) create mode 100644 docs/my-website/docs/providers/modelscope.md create mode 100644 litellm/llms/modelscope/chat/transformation.py create mode 100644 tests/test_litellm/llms/modelscope/test_modelscope_chat_transformation.py diff --git a/docs/my-website/docs/providers/modelscope.md b/docs/my-website/docs/providers/modelscope.md new file mode 100644 index 00000000000..b88041ae176 --- /dev/null +++ b/docs/my-website/docs/providers/modelscope.md @@ -0,0 +1,242 @@ +# ModelScope +LiteLLM supports running inference across multiple services for models hosted on the ModelScope Hub. + +## Supported Models + +### Serverless Inference Providers +You can check available models for an inference provider by going to [modelscope.cn/models](https://modelscope.cn/models), clicking the "API-Inference" and the "Other" filter tab, and selecting your desired provider. + +For example, you can find all Qwen3 series models [here](https://modelscope.cn/models?filter=inference_type&model_type=qwen3&page=1&tabKey=other). + + +### Dedicated Inference Endpoints +Refer to the [Inference Endpoints catalog](https://modelscope.cn/models?filter=inference_type&page=1&tabKey=task) for a list of available models. + +## Usage + + +### Authentication +With a single ModelScope token, you can access inference through multiple providers. Your calls are routed through ModelScope and the usage is free. +However, please ensure you bind your Alibaba Cloud account before use. For details, refer to the following two links. +- [API-Infereference](https://modelscope.cn/docs/model-service/API-Inference/intro) +- [Binding Alibaba Cloud Account](https://modelscope.cn/docs/accounts/aliyun-binding-and-authorization) + +Simply set the `MODELSCOPE_TOKEN` environment variable with your ModelScope token, you can create one here: https://modelscope.cn/my/myaccesstoken. + +```bash +export MODELSCOPE_TOKEN="123xxxxxx" +``` +or alternatively, you can pass your ModelScope token as a parameter: +```python +completion(..., api_key="123xxxxxx") +``` + +### Getting Started + +To use a ModelScope model, specify the model you want to use in the following format: +``` +// +``` +Where `/` is the ModelScope model ID. + +Examples: + +```python +# Run Llama-4-Scout-17B-16E-Instruct inference through LLM-Research +completion(model="modelscope/LLM-Research/Llama-4-Scout-17B-16E-Instruct",...) + +# Run DeepSeek-R1 inference through DeepSeek +completion(model="modelscope/deepseek-ai/DeepSeek-R1-0528",...) + +# Run Qwen3-8B inference through Qwen +completion(model="modelscope/Qwen/Qwen3-8B",...) +``` + + +### Basic Completion +Here's an example of chat completion using the Qwen3-8B model through Qwen: + +```python +import os +from litellm import completion + +os.environ["MODELSCOPE_TOKEN"] = "123xxxxxx" + +response = completion( + model="modelscope/Qwen/Qwen3-Coder-480B-A35B-Instruct", + messages=[ + { + "role": "user", + "content": "How many r's are in the word 'strawberry'?", + } + ], +) +print(response) +``` + +### Streaming +Now, let's see what a streaming request looks like. + +```python +import os +from litellm import completion + +os.environ["MODELSCOPE_TOKEN"] = "123xxxxxx" + +response = completion( + model="modelscope/Qwen/Qwen3-Coder-480B-A35B-Instruct", + messages=[ + { + "role": "user", + "content": "How many r's are in the word `strawberry`?", + + } + ], + stream=True, +) + +for chunk in response: + print(chunk) +``` + +### Image Input +You can also pass images when the model supports it. Here is an example using [Qwen/Qwen2.5-VL-72B-Instruct](https://modelscope.cn/models/Qwen/Qwen2.5-VL-72B-Instruct) model. + +```python +from litellm import completion + +# Set your ModelScope Token +os.environ["MODELSCOPE_TOKEN"] = "123xxxxxx" + +messages=[ + { + "role": "user", + "content": [ + {"type": "text", "text": "What's in this image?"}, + { + "type": "image_url", + "image_url": { + "url": "https://modelscope.oss-cn-beijing.aliyuncs.com/demo/images/audrey_hepburn.jpg" + } + } + ] + } + ] + +response = completion( + model="modelscope/Qwen/Qwen2.5-VL-72B-Instruct", + messages=messages, +) +print(response.choices[0]) +``` + +## Model Deployment +SwingDeploy deployment service is a one-stop model deployment solution launched by ModelScope, aiming to provide developers with end-to-end services from model selection to cloud deployment. Through standardized deployment processes and cloud resource adaptation capabilities, users can quickly deploy the rich models of the Moda community (in multiple fields such as voice, video, and NLP) to the target cloud environment, achieving efficient implementation of model inference services. You can refer to [SwingDeploy](https://modelscope.cn/docs/model-service/deployment/intro) for more details. + +## LiteLLM Proxy Server with ModelScope models +You can set up a [LiteLLM Proxy Server](https://docs.litellm.ai/#litellm-proxy-server-llm-gateway) to serve ModelScope models through any of the supported Inference Providers. Here's how to do it: + +### Step 1. Setup the config file + +In this case, we are configuring a proxy to serve `Qwen3-Coder-480B-A35B-Instruct` from ModelScope. + +```yaml +model_list: + - model_name: my-model + litellm_params: + model: modelscope/Qwen/Qwen3-Coder-480B-A35B-Instruct + api_key: os.environ/MODELSCOPE_TOKEN # ensure you have `MODELSCOPE_TOKEN` in your .env +``` + +### Step 2. Start the server +```bash +litellm --config /path/to/config.yaml +``` + +### Step 3. Make a request to the server + + + +```shell +curl --location 'http://0.0.0.0:4000/chat/completions' \ + --header 'Content-Type: application/json' \ + --data '{ + "model": "my-model", + "messages": [ + { + "role": "user", + "content": "Hello, how are you?" + } + ] +}' +``` + + +## Usage with LiteLLM Proxy Server + +Here's how to call a ModelScope model with the LiteLLM Proxy Server + +1. Modify the config.yaml + + ```yaml showLineNumbers + model_list: + - model_name: my-model + litellm_params: + model: modelscope// # add modelscope/ prefix to route as ModelScope provider + api_key: api-key # api key to send your model + ``` + + +2. Start the proxy + + ```bash + $ litellm --config /path/to/config.yaml + ``` + +3. Send Request to LiteLLM Proxy Server + + + + + + ```python showLineNumbers + import openai + client = openai.OpenAI( + api_key="123xxxx", # pass litellm proxy key, if you're using virtual keys + base_url="http://0.0.0.0:4000" # litellm-proxy-base url + ) + + response = client.chat.completions.create( + model="my-model", + messages = [ + { + "role": "user", + "content": "what llm are you" + } + ], + ) + + print(response) + ``` + + + + + ```shell + curl --location 'http://0.0.0.0:4000/chat/completions' \ + --header 'Authorization: Bearer 1234xxxxx' \ + --header 'Content-Type: application/json' \ + --data '{ + "model": "my-model", + "messages": [ + { + "role": "user", + "content": "what llm are you" + } + ] + }' + ``` + + + + diff --git a/litellm/__init__.py b/litellm/__init__.py index c954f5fd31e..405a83eea92 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -625,6 +625,7 @@ aiml_models: Set = set() deepgram_models: Set = set() elevenlabs_models: Set = set() dashscope_models: Set = set() +modelscope_models: Set = set() moonshot_models: Set = set() publicai_models: Set = set() v0_models: Set = set() @@ -872,7 +873,9 @@ def add_known_models(model_cost_map: Optional[Dict] = None): elif value.get("litellm_provider") == "heroku": heroku_models.add(key) elif value.get("litellm_provider") == "dashscope": - dashscope_models.add(key) + dashscope_models.append(key) + elif value.get("litellm_provider") == "modelscope": + modelscope_models.append(key) elif value.get("litellm_provider") == "moonshot": moonshot_models.add(key) elif value.get("litellm_provider") == "publicai": @@ -992,6 +995,7 @@ model_list = list( | zai_models | fal_ai_models | deepseek_models + | modelscope_models | azure_ai_models | voyage_models | infinity_models @@ -1123,6 +1127,7 @@ models_by_provider: dict = { "elevenlabs": elevenlabs_models, "heroku": heroku_models, "dashscope": dashscope_models, + "modelscope": modelscope_models, "moonshot": moonshot_models, "publicai": publicai_models, "v0": v0_models, @@ -1234,14 +1239,30 @@ from .llms.topaz.common_utils import TopazModelInfo # OpenAIGPTConfig, OpenAIGPT5Config, etc. are lazy loaded - instances will be created on first access from .llms.xai.common_utils import XAIModelInfo -# PublicAI now uses JSON-based configuration (see litellm/llms/openai_like/providers.json) -# All remaining configs are now lazy loaded - see _lazy_imports_registry.py - -# Import LlmProviders here (before main import) because it's imported during import time -# in multiple places including openai.py (via main import) -from litellm.types.utils import LlmProviders - -## Lazy loading this is not straightforward, will leave it here for now. +from .llms.azure.chat.gpt_transformation import AzureOpenAIConfig +from .llms.azure.completion.transformation import AzureOpenAITextConfig +from .llms.hosted_vllm.chat.transformation import HostedVLLMChatConfig +from .llms.llamafile.chat.transformation import LlamafileChatConfig +from .llms.litellm_proxy.chat.transformation import LiteLLMProxyChatConfig +from .llms.vllm.completion.transformation import VLLMConfig +from .llms.deepseek.chat.transformation import DeepSeekChatConfig +from .llms.lm_studio.chat.transformation import LMStudioChatConfig +from .llms.lm_studio.embed.transformation import LmStudioEmbeddingConfig +from .llms.nscale.chat.transformation import NscaleConfig +from .llms.perplexity.chat.transformation import PerplexityChatConfig +from .llms.azure.chat.o_series_transformation import AzureOpenAIO1Config +from .llms.watsonx.completion.transformation import IBMWatsonXAIConfig +from .llms.watsonx.chat.transformation import IBMWatsonXChatConfig +from .llms.watsonx.embed.transformation import IBMWatsonXEmbeddingConfig +from .llms.github_copilot.chat.transformation import GithubCopilotConfig +from .llms.nebius.chat.transformation import NebiusConfig +from .llms.dashscope.chat.transformation import DashScopeChatConfig +from .llms.modelscope.chat.transformation import ModelScopeChatConfig +from .llms.moonshot.chat.transformation import MoonshotChatConfig +from .llms.v0.chat.transformation import V0ChatConfig +from .llms.morph.chat.transformation import MorphChatConfig +from .llms.lambda_ai.chat.transformation import LambdaAIChatConfig +from .llms.hyperbolic.chat.transformation import HyperbolicChatConfig from .main import * # type: ignore from .compression import compress # type: ignore[no-redef] diff --git a/litellm/constants.py b/litellm/constants.py index 26e25d0cef3..3a6a6c8a31b 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -614,6 +614,7 @@ LITELLM_CHAT_PROVIDERS = [ "nscale", "nebius", "dashscope", + "modelscope", "moonshot", "publicai", "v0", @@ -770,6 +771,7 @@ openai_compatible_endpoints: List = [ "inference.api.nscale.com/v1", "api.studio.nebius.ai/v1", "https://dashscope-intl.aliyuncs.com/compatible-mode/v1", + "https://api-inference.modelscope.cn/v1", "https://api.moonshot.ai/v1", "https://api.publicai.co/v1", "https://api.synthetic.new/openai/v1", @@ -833,6 +835,7 @@ openai_compatible_providers: List = [ "nscale", "nebius", "dashscope", + "modelscope", "moonshot", "v0", "helicone", @@ -858,6 +861,7 @@ openai_text_completion_compatible_providers: List = ( "featherless_ai", "nebius", "dashscope", + "modelscope", "moonshot", "publicai", "synthetic", @@ -1081,42 +1085,56 @@ dashscope_models: set = set( ] ) -nebius_embedding_models: set = set( - [ - "BAAI/bge-en-icl", - "BAAI/bge-multilingual-gemma2", - "intfloat/e5-mistral-7b-instruct", - ] -) +modelscope_models: List = [ + "LLM-Research/c4ai-command-r-plus-08-2024", + "mistralai/Mistral-Small-Instruct-2409", + "mistralai/Ministral-8B-Instruct-2410", + "mistralai/Mistral-Large-Instruct-2407", + "Qwen/Qwen2.5-Coder-32B-Instruct", + "Qwen/Qwen2.5-Coder-14B-Instruct", + "Qwen/Qwen2.5-Coder-7B-Instruct", + "Qwen/Qwen2.5-72B-Instruct", + "Qwen/Qwen2.5-32B-Instruct", + "Qwen/Qwen2.5-14B-Instruct", + "Qwen/Qwen2.5-7B-Instruct", + "Qwen/QwQ-32B-Preview", + "opencompass/CompassJudger-1-32B-Instruct", + "Qwen/QVQ-72B-Preview", + "Qwen/Qwen2-VL-7B-Instruct", + "Qwen/Qwen2.5-14B-Instruct-1M", + "Qwen/Qwen2.5-7B-Instruct-1M", + "Qwen/Qwen2.5-VL-3B-Instruct", + "Qwen/Qwen2.5-VL-7B-Instruct", + "Qwen/Qwen2.5-VL-72B-Instruct", + "deepseek-ai/DeepSeek-V3", + "Qwen/QwQ-32B", + "XGenerationLab/XiYanSQL-QwenCoder-32B-2412", + "Qwen/Qwen2.5-VL-32B-Instruct", + "LLM-Research/Llama-4-Scout-17B-16E-Instruct", + "LLM-Research/Llama-4-Maverick-17B-128E-Instruct", + "Qwen/Qwen3-0.6B", + "Qwen/Qwen3-1.7B", + "Qwen/Qwen3-4B", + "Qwen/Qwen3-8B", + "Qwen/Qwen3-14B", + "Qwen/Qwen3-30B-A3B", + "Qwen/Qwen3-32B", + "Qwen/Qwen3-235B-A22B", + "deepseek-ai/DeepSeek-R1-0528", + "MiniMax/MiniMax-M1-80k", + "Qwen/Qwen3-235B-A22B-Instruct-2507", + "Qwen/Qwen3-Coder-480B-A35B-Instruct", + "Qwen/Qwen3-235B-A22B-Thinking-2507", + "ZhipuAI/GLM-4.5", + "Qwen/Qwen3-30B-A3B-Thinking-2507", + "Qwen/Qwen3-Coder-30B-A3B-Instruct", +] -WANDB_MODELS: set = set( - [ - # openai models - "openai/gpt-oss-120b", - "openai/gpt-oss-20b", - # zai-org models - "zai-org/GLM-4.5", - # Qwen models - "Qwen/Qwen3-235B-A22B-Instruct-2507", - "Qwen/Qwen3-Coder-480B-A35B-Instruct", - "Qwen/Qwen3-235B-A22B-Thinking-2507", - # moonshotai - "moonshotai/Kimi-K2-Instruct", - "moonshotai/Kimi-K2.5", - # MiniMaxAI - "MiniMaxAI/MiniMax-M2.5", - # meta models - "meta-llama/Llama-3.1-8B-Instruct", - "meta-llama/Llama-3.3-70B-Instruct", - "meta-llama/Llama-4-Scout-17B-16E-Instruct", - # deepseek-ai - "deepseek-ai/DeepSeek-V3.1", - "deepseek-ai/DeepSeek-R1-0528", - "deepseek-ai/DeepSeek-V3-0324", - # microsoft - "microsoft/Phi-4-mini-instruct", - ] -) +nebius_embedding_models: List = [ + "BAAI/bge-en-icl", + "BAAI/bge-multilingual-gemma2", + "intfloat/e5-mistral-7b-instruct", +] BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[ "cohere", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index a71000f00f8..7b9e575e030 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -334,6 +334,9 @@ def get_llm_provider( # noqa: PLR0915 elif endpoint == "dashscope-intl.aliyuncs.com/compatible-mode/v1": custom_llm_provider = "dashscope" dynamic_api_key = get_secret_str("DASHSCOPE_API_KEY") + elif endpoint == "api-inference.modelscope.cn/v1": + custom_llm_provider = "modelscope" + dynamic_api_key = get_secret_str("MODELSCOPE_API_KEY") elif endpoint == "api.moonshot.ai/v1": custom_llm_provider = "moonshot" dynamic_api_key = get_secret_str("MOONSHOT_API_KEY") @@ -921,6 +924,13 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 ) = litellm.DashScopeChatConfig()._get_openai_compatible_provider_info( api_base, api_key ) + elif custom_llm_provider == "modelscope": + ( + api_base, + dynamic_api_key, + ) = litellm.ModelScopeChatConfig()._get_openai_compatible_provider_info( + api_base, api_key + ) elif custom_llm_provider == "moonshot": ( api_base, diff --git a/litellm/llms/modelscope/chat/transformation.py b/litellm/llms/modelscope/chat/transformation.py new file mode 100644 index 00000000000..b657bc8950f --- /dev/null +++ b/litellm/llms/modelscope/chat/transformation.py @@ -0,0 +1,75 @@ +""" +Translates from OpenAI's `/v1/chat/completions` to ModelScope's `/v1/chat/completions` +""" + +from typing import Any, Coroutine, List, Literal, Optional, Tuple, Union, overload + +from litellm.litellm_core_utils.prompt_templates.common_utils import ( + handle_messages_with_content_list_to_str_conversion, +) +from litellm.secret_managers.main import get_secret_str +from litellm.types.llms.openai import AllMessageValues + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig + + +class ModelScopeChatConfig(OpenAIGPTConfig): + @overload + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: Literal[True] + ) -> Coroutine[Any, Any, List[AllMessageValues]]: ... + + @overload + def _transform_messages( + self, + messages: List[AllMessageValues], + model: str, + is_async: Literal[False] = False, + ) -> List[AllMessageValues]: ... + + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: bool = False + ) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]: + """ + ModelScope does not support content in list format. + """ + messages = handle_messages_with_content_list_to_str_conversion(messages) + if is_async: + return super()._transform_messages( + messages=messages, model=model, is_async=True + ) + else: + return super()._transform_messages( + messages=messages, model=model, is_async=False + ) + + def _get_openai_compatible_provider_info( + self, api_base: Optional[str], api_key: Optional[str] + ) -> Tuple[Optional[str], Optional[str]]: + api_base = ( + api_base + or get_secret_str("MODELSCOPE_API_BASE") + or "https://api-inference.modelscope.cn/v1" + ) # type: ignore + dynamic_api_key = api_key or get_secret_str("MODELSCOPE_API_KEY") + return api_base, dynamic_api_key + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + """ + If api_base is not provided, use the default ModelScope /chat/completions endpoint. + """ + if not api_base: + api_base = "https://api-inference.modelscope.cn/v1" + + if not api_base.endswith("/chat/completions"): + api_base = f"{api_base}/chat/completions" + + return api_base diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 3dcff2be689..5dc0ece37fe 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3296,6 +3296,7 @@ class LlmProviders(str, Enum): CODESTRAL = "codestral" TEXT_COMPLETION_CODESTRAL = "text-completion-codestral" DASHSCOPE = "dashscope" + MODELSCOPE = "modelscope" MOONSHOT = "moonshot" PUBLICAI = "publicai" V0 = "v0" diff --git a/litellm/utils.py b/litellm/utils.py index 7cac830b2c2..8790feadae0 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -6729,6 +6729,11 @@ def validate_environment( # noqa: PLR0915 keys_in_environment = True else: missing_keys.append("DASHSCOPE_API_KEY") + elif custom_llm_provider == "modelscope": + if "MODELSCOPE_API_KEY" in os.environ: + keys_in_environment = True + else: + missing_keys.append("MODELSCOPE_API_KEY") elif custom_llm_provider == "moonshot": if "MOONSHOT_API_KEY" in os.environ: keys_in_environment = True @@ -8400,6 +8405,7 @@ class ProviderConfigManager: LlmProviders.NEBIUS: (lambda: litellm.NebiusConfig(), False), LlmProviders.WANDB: (lambda: litellm.WandbConfig(), False), LlmProviders.DASHSCOPE: (lambda: litellm.DashScopeChatConfig(), False), + LlmProviders.MODELSCOPE: (lambda: litellm.ModelScopeChatConfig(), False), LlmProviders.MOONSHOT: (lambda: litellm.MoonshotChatConfig(), False), LlmProviders.DOCKER_MODEL_RUNNER: ( lambda: litellm.DockerModelRunnerChatConfig(), diff --git a/tests/test_litellm/llms/modelscope/test_modelscope_chat_transformation.py b/tests/test_litellm/llms/modelscope/test_modelscope_chat_transformation.py new file mode 100644 index 00000000000..7e233922189 --- /dev/null +++ b/tests/test_litellm/llms/modelscope/test_modelscope_chat_transformation.py @@ -0,0 +1,114 @@ +""" +Unit tests for ModelScope configuration. + +These tests validate the DashScopeConfig class which extends OpenAIGPTConfig. +ModelScope is an OpenAI-compatible provider with minor customizations. +""" + +import os +import sys + +sys.path.insert( + 0, os.path.abspath("../../../../..") +) # Adds the parent directory to the system path + +import pytest + +import litellm +from litellm import completion +from litellm.llms.modelscope.chat.transformation import ModelScopeChatConfig + + +class TestModelScopeConfig: + """Test class for ModelScope functionality""" + + def test_default_api_base(self): + """Test that default API base is used when none is provided""" + config = ModelScopeChatConfig() + headers = {} + api_key = "fake-modelscope-key" + + # Call validate_environment without specifying api_base + result = config.validate_environment( + headers=headers, + model="Qwen/Qwen3-8B", + messages=[{"role": "user", "content": "Hey"}], + optional_params={}, + litellm_params={}, + api_key=api_key, + api_base=None, # Not providing api_base + ) + + # Verify headers are still set correctly + assert result["Authorization"] == f"Bearer {api_key}" + assert result["Content-Type"] == "application/json" + + # We can't directly test the api_base value here since validate_environment + # only returns the headers, but we can verify it doesn't raise an exception + # which would happen if api_base handling was incorrect + + @pytest.mark.respx() + def test_modelscope_completion_mock(self, respx_mock): + """ + Mock test for ModelScope completion using the model format from docs. + This test mocks the actual HTTP request to test the integration properly. + """ + + litellm.disable_aiohttp_transport = ( + True # since this uses respx, we need to set use_aiohttp_transport to False + ) + + # Set up environment variables for the test + api_key = "fake-modelscope-key" + api_base = "https://api-inference.modelscope.cn/v1" + model = "Qwen/Qwen3-8B" + model_name = "Qwen3-8B" # The actual model name without provider prefix + + # Mock the HTTP request to the ModelScope API + respx_mock.post(f"{api_base}/chat/completions").respond( + json={ + "id": "chatcmpl-123", + "object": "chat.completion", + "created": 1677652288, + "model": model_name, + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": '```python\nprint("Hey from LiteLLM!")\n```\n\nThis simple Python code prints a greeting message from LiteLLM.', + }, + "finish_reason": "stop", + } + ], + "usage": { + "prompt_tokens": 9, + "completion_tokens": 12, + "total_tokens": 21, + }, + }, + status_code=200, + ) + + # Make the actual API call through LiteLLM + response = completion( + model=model, + messages=[ + {"role": "user", "content": "write code for saying hey from LiteLLM"} + ], + api_key=api_key, + api_base=api_base, + ) + + # Verify response structure + assert response is not None + assert hasattr(response, "choices") + assert len(response.choices) > 0 + assert hasattr(response.choices[0], "message") + assert hasattr(response.choices[0].message, "content") + assert response.choices[0].message.content is not None + + # Check for specific content in the response + assert "```python" in response.choices[0].message.content + assert "Hey from LiteLLM" in response.choices[0].message.content +