From 84bc74a4fd540053f34731a5fefdd8c9d1d1a7f2 Mon Sep 17 00:00:00 2001 From: yrk <2493404415@qq.com> Date: Fri, 15 May 2026 09:49:35 +0800 Subject: [PATCH] update modelscope model list --- docs/my-website/docs/providers/modelscope.md | 242 ------------------- litellm/constants.py | 68 +++--- 2 files changed, 29 insertions(+), 281 deletions(-) delete mode 100644 docs/my-website/docs/providers/modelscope.md diff --git a/docs/my-website/docs/providers/modelscope.md b/docs/my-website/docs/providers/modelscope.md deleted file mode 100644 index b88041ae176..00000000000 --- a/docs/my-website/docs/providers/modelscope.md +++ /dev/null @@ -1,242 +0,0 @@ -# ModelScope -LiteLLM supports running inference across multiple services for models hosted on the ModelScope Hub. - -## Supported Models - -### Serverless Inference Providers -You can check available models for an inference provider by going to [modelscope.cn/models](https://modelscope.cn/models), clicking the "API-Inference" and the "Other" filter tab, and selecting your desired provider. - -For example, you can find all Qwen3 series models [here](https://modelscope.cn/models?filter=inference_type&model_type=qwen3&page=1&tabKey=other). - - -### Dedicated Inference Endpoints -Refer to the [Inference Endpoints catalog](https://modelscope.cn/models?filter=inference_type&page=1&tabKey=task) for a list of available models. - -## Usage - - -### Authentication -With a single ModelScope token, you can access inference through multiple providers. Your calls are routed through ModelScope and the usage is free. -However, please ensure you bind your Alibaba Cloud account before use. For details, refer to the following two links. -- [API-Infereference](https://modelscope.cn/docs/model-service/API-Inference/intro) -- [Binding Alibaba Cloud Account](https://modelscope.cn/docs/accounts/aliyun-binding-and-authorization) - -Simply set the `MODELSCOPE_TOKEN` environment variable with your ModelScope token, you can create one here: https://modelscope.cn/my/myaccesstoken. - -```bash -export MODELSCOPE_TOKEN="123xxxxxx" -``` -or alternatively, you can pass your ModelScope token as a parameter: -```python -completion(..., api_key="123xxxxxx") -``` - -### Getting Started - -To use a ModelScope model, specify the model you want to use in the following format: -``` -// -``` -Where `/` is the ModelScope model ID. - -Examples: - -```python -# Run Llama-4-Scout-17B-16E-Instruct inference through LLM-Research -completion(model="modelscope/LLM-Research/Llama-4-Scout-17B-16E-Instruct",...) - -# Run DeepSeek-R1 inference through DeepSeek -completion(model="modelscope/deepseek-ai/DeepSeek-R1-0528",...) - -# Run Qwen3-8B inference through Qwen -completion(model="modelscope/Qwen/Qwen3-8B",...) -``` - - -### Basic Completion -Here's an example of chat completion using the Qwen3-8B model through Qwen: - -```python -import os -from litellm import completion - -os.environ["MODELSCOPE_TOKEN"] = "123xxxxxx" - -response = completion( - model="modelscope/Qwen/Qwen3-Coder-480B-A35B-Instruct", - messages=[ - { - "role": "user", - "content": "How many r's are in the word 'strawberry'?", - } - ], -) -print(response) -``` - -### Streaming -Now, let's see what a streaming request looks like. - -```python -import os -from litellm import completion - -os.environ["MODELSCOPE_TOKEN"] = "123xxxxxx" - -response = completion( - model="modelscope/Qwen/Qwen3-Coder-480B-A35B-Instruct", - messages=[ - { - "role": "user", - "content": "How many r's are in the word `strawberry`?", - - } - ], - stream=True, -) - -for chunk in response: - print(chunk) -``` - -### Image Input -You can also pass images when the model supports it. Here is an example using [Qwen/Qwen2.5-VL-72B-Instruct](https://modelscope.cn/models/Qwen/Qwen2.5-VL-72B-Instruct) model. - -```python -from litellm import completion - -# Set your ModelScope Token -os.environ["MODELSCOPE_TOKEN"] = "123xxxxxx" - -messages=[ - { - "role": "user", - "content": [ - {"type": "text", "text": "What's in this image?"}, - { - "type": "image_url", - "image_url": { - "url": "https://modelscope.oss-cn-beijing.aliyuncs.com/demo/images/audrey_hepburn.jpg" - } - } - ] - } - ] - -response = completion( - model="modelscope/Qwen/Qwen2.5-VL-72B-Instruct", - messages=messages, -) -print(response.choices[0]) -``` - -## Model Deployment -SwingDeploy deployment service is a one-stop model deployment solution launched by ModelScope, aiming to provide developers with end-to-end services from model selection to cloud deployment. Through standardized deployment processes and cloud resource adaptation capabilities, users can quickly deploy the rich models of the Moda community (in multiple fields such as voice, video, and NLP) to the target cloud environment, achieving efficient implementation of model inference services. You can refer to [SwingDeploy](https://modelscope.cn/docs/model-service/deployment/intro) for more details. - -## LiteLLM Proxy Server with ModelScope models -You can set up a [LiteLLM Proxy Server](https://docs.litellm.ai/#litellm-proxy-server-llm-gateway) to serve ModelScope models through any of the supported Inference Providers. Here's how to do it: - -### Step 1. Setup the config file - -In this case, we are configuring a proxy to serve `Qwen3-Coder-480B-A35B-Instruct` from ModelScope. - -```yaml -model_list: - - model_name: my-model - litellm_params: - model: modelscope/Qwen/Qwen3-Coder-480B-A35B-Instruct - api_key: os.environ/MODELSCOPE_TOKEN # ensure you have `MODELSCOPE_TOKEN` in your .env -``` - -### Step 2. Start the server -```bash -litellm --config /path/to/config.yaml -``` - -### Step 3. Make a request to the server - - - -```shell -curl --location 'http://0.0.0.0:4000/chat/completions' \ - --header 'Content-Type: application/json' \ - --data '{ - "model": "my-model", - "messages": [ - { - "role": "user", - "content": "Hello, how are you?" - } - ] -}' -``` - - -## Usage with LiteLLM Proxy Server - -Here's how to call a ModelScope model with the LiteLLM Proxy Server - -1. Modify the config.yaml - - ```yaml showLineNumbers - model_list: - - model_name: my-model - litellm_params: - model: modelscope// # add modelscope/ prefix to route as ModelScope provider - api_key: api-key # api key to send your model - ``` - - -2. Start the proxy - - ```bash - $ litellm --config /path/to/config.yaml - ``` - -3. Send Request to LiteLLM Proxy Server - - - - - - ```python showLineNumbers - import openai - client = openai.OpenAI( - api_key="123xxxx", # pass litellm proxy key, if you're using virtual keys - base_url="http://0.0.0.0:4000" # litellm-proxy-base url - ) - - response = client.chat.completions.create( - model="my-model", - messages = [ - { - "role": "user", - "content": "what llm are you" - } - ], - ) - - print(response) - ``` - - - - - ```shell - curl --location 'http://0.0.0.0:4000/chat/completions' \ - --header 'Authorization: Bearer 1234xxxxx' \ - --header 'Content-Type: application/json' \ - --data '{ - "model": "my-model", - "messages": [ - { - "role": "user", - "content": "what llm are you" - } - ] - }' - ``` - - - - diff --git a/litellm/constants.py b/litellm/constants.py index 3c439f9a36d..cb8414870ad 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1085,6 +1085,12 @@ dashscope_models: set = set( ] ) +nebius_embedding_models: List = [ + "BAAI/bge-en-icl", + "BAAI/bge-multilingual-gemma2", + "intfloat/e5-mistral-7b-instruct", +] + WANDB_MODELS: set = set( [ # openai models @@ -1115,15 +1121,7 @@ WANDB_MODELS: set = set( ) modelscope_models: List = [ - # LLM-Research models - "LLM-Research/c4ai-command-r-plus-08-2024", - "LLM-Research/Llama-4-Scout-17B-16E-Instruct", - "LLM-Research/Llama-4-Maverick-17B-128E-Instruct", - # Mistral models - "mistralai/Mistral-Small-Instruct-2409", - "mistralai/Ministral-8B-Instruct-2410", - "mistralai/Mistral-Large-Instruct-2407", - # Qwen3 series + # Qwen series models "Qwen/Qwen3-0.6B", "Qwen/Qwen3-1.7B", "Qwen/Qwen3-4B", @@ -1133,41 +1131,33 @@ modelscope_models: List = [ "Qwen/Qwen3-32B", "Qwen/Qwen3-235B-A22B", "Qwen/Qwen3-235B-A22B-Instruct-2507", - "Qwen/Qwen3-Coder-480B-A35B-Instruct", "Qwen/Qwen3-235B-A22B-Thinking-2507", "Qwen/Qwen3-30B-A3B-Thinking-2507", "Qwen/Qwen3-Coder-30B-A3B-Instruct", - # Qwen3.5 series (new) - "Qwen/Qwen3.5-0.6B", - "Qwen/Qwen3.5-1.7B", - "Qwen/Qwen3.5-3B", - "Qwen/Qwen3.5-7B", - "Qwen/Qwen3.5-14B", - "Qwen/Qwen3.5-30B-A3B", - "Qwen/Qwen3.5-32B", - "Qwen/Qwen3.5-235B-A22B", - "Qwen/Qwen3.5-235B-A22B-Instruct-2507", - "Qwen/Qwen3.5-Coder-480B-A35B-Instruct", - "Qwen/Qwen3.5-235B-A22B-Thinking-2507", - "Qwen/Qwen3.5-30B-A3B-Thinking-2507", - "Qwen/Qwen3.5-Coder-30B-A3B-Instruct", - # Other models - "Qwen/QwQ-32B-Preview", - "opencompass/CompassJudger-1-32B-Instruct", - "Qwen/QVQ-72B-Preview", - "Qwen/Qwen2-VL-7B-Instruct", - "deepseek-ai/DeepSeek-V3", + "Qwen/Qwen3-Coder-480B-A35B-Instruct", + "Qwen/Qwen3-Next-80B-A3B-Instruct", + "Qwen/Qwen3-Next-80B-A3B-Thinking", + "Qwen/Qwen3-VL-235B-A22B-Instruct", + "Qwen/Qwen3-VL-8B-Instruct", + "Qwen/Qwen3-VL-8B-Thinking", + "Qwen/Qwen3.5-122B-A10B", + "Qwen/Qwen3.5-27B", + "Qwen/Qwen3.5-35B-A3B", + "Qwen/Qwen3.5-397B-A17B", "Qwen/QwQ-32B", - "XGenerationLab/XiYanSQL-QwenCoder-32B-2412", + "Qwen/QwQ-32B-Preview", + "Qwen/QVQ-72B-Preview", + "Qwen/Qwen-Image-Edit", + # DeepSeek series models "deepseek-ai/DeepSeek-R1-0528", - "MiniMax/MiniMax-M1-80k", - "ZhipuAI/GLM-4.5", -] - -nebius_embedding_models: List = [ - "BAAI/bge-en-icl", - "BAAI/bge-multilingual-gemma2", - "intfloat/e5-mistral-7b-instruct", + "deepseek-ai/DeepSeek-R1-Distill-Llama-70B", + "deepseek-ai/DeepSeek-R1-Distill-Llama-8B", + "deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B", + "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B", + "deepseek-ai/DeepSeek-R1-Distill-Qwen-32B", + "deepseek-ai/DeepSeek-R1-Distill-Qwen-7B", + "deepseek-ai/DeepSeek-V3.2", + "deepseek-ai/DeepSeek-V4-Flash", ] BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[