add modelscope api support

This commit is contained in:
yrk 2026-05-14 14:59:41 +08:00
parent 97894f1603
commit f6ce4e16f8
4 changed files with 78 additions and 52 deletions

View file

@ -873,9 +873,9 @@ def add_known_models(model_cost_map: Optional[Dict] = None):
elif value.get("litellm_provider") == "heroku":
heroku_models.add(key)
elif value.get("litellm_provider") == "dashscope":
dashscope_models.append(key)
dashscope_models.add(key)
elif value.get("litellm_provider") == "modelscope":
modelscope_models.append(key)
modelscope_models.add(key)
elif value.get("litellm_provider") == "moonshot":
moonshot_models.add(key)
elif value.get("litellm_provider") == "publicai":
@ -1239,30 +1239,14 @@ from .llms.topaz.common_utils import TopazModelInfo
# OpenAIGPTConfig, OpenAIGPT5Config, etc. are lazy loaded - instances will be created on first access
from .llms.xai.common_utils import XAIModelInfo
from .llms.azure.chat.gpt_transformation import AzureOpenAIConfig
from .llms.azure.completion.transformation import AzureOpenAITextConfig
from .llms.hosted_vllm.chat.transformation import HostedVLLMChatConfig
from .llms.llamafile.chat.transformation import LlamafileChatConfig
from .llms.litellm_proxy.chat.transformation import LiteLLMProxyChatConfig
from .llms.vllm.completion.transformation import VLLMConfig
from .llms.deepseek.chat.transformation import DeepSeekChatConfig
from .llms.lm_studio.chat.transformation import LMStudioChatConfig
from .llms.lm_studio.embed.transformation import LmStudioEmbeddingConfig
from .llms.nscale.chat.transformation import NscaleConfig
from .llms.perplexity.chat.transformation import PerplexityChatConfig
from .llms.azure.chat.o_series_transformation import AzureOpenAIO1Config
from .llms.watsonx.completion.transformation import IBMWatsonXAIConfig
from .llms.watsonx.chat.transformation import IBMWatsonXChatConfig
from .llms.watsonx.embed.transformation import IBMWatsonXEmbeddingConfig
from .llms.github_copilot.chat.transformation import GithubCopilotConfig
from .llms.nebius.chat.transformation import NebiusConfig
from .llms.dashscope.chat.transformation import DashScopeChatConfig
from .llms.modelscope.chat.transformation import ModelScopeChatConfig
from .llms.moonshot.chat.transformation import MoonshotChatConfig
from .llms.v0.chat.transformation import V0ChatConfig
from .llms.morph.chat.transformation import MorphChatConfig
from .llms.lambda_ai.chat.transformation import LambdaAIChatConfig
from .llms.hyperbolic.chat.transformation import HyperbolicChatConfig
# PublicAI now uses JSON-based configuration (see litellm/llms/openai_like/providers.json)
# All remaining configs are now lazy loaded - see _lazy_imports_registry.py
# Import LlmProviders here (before main import) because it's imported during import time
# in multiple places including openai.py (via main import)
from litellm.types.utils import LlmProviders
## Lazy loading this is not straightforward, will leave it here for now.
from .main import * # type: ignore
from .compression import compress # type: ignore[no-redef]
@ -1958,6 +1942,9 @@ if TYPE_CHECKING:
from .llms.dashscope.rerank.transformation import (
DashScopeRerankConfig as DashScopeRerankConfig,
)
from .llms.modelscope.chat.transformation import (
ModelScopeChatConfig as ModelScopeChatConfig,
)
from .llms.moonshot.chat.transformation import (
MoonshotChatConfig as MoonshotChatConfig,
)

View file

@ -304,6 +304,7 @@ LLM_CONFIG_NAMES = (
"GigaChatConfig",
"GigaChatEmbeddingConfig",
"DashScopeChatConfig",
"ModelScopeChatConfig",
"MoonshotChatConfig",
"DockerModelRunnerChatConfig",
"V0ChatConfig",
@ -1150,6 +1151,10 @@ _LLM_CONFIGS_IMPORT_MAP = {
".llms.dashscope.chat.transformation",
"DashScopeChatConfig",
),
"ModelScopeChatConfig": (
".llms.modelscope.chat.transformation",
"ModelScopeChatConfig",
),
"MoonshotChatConfig": (".llms.moonshot.chat.transformation", "MoonshotChatConfig"),
"DockerModelRunnerChatConfig": (
".llms.docker_model_runner.chat.transformation",

View file

@ -1085,33 +1085,45 @@ dashscope_models: set = set(
]
)
WANDB_MODELS: set = set(
[
# openai models
"openai/gpt-oss-120b",
"openai/gpt-oss-20b",
# zai-org models
"zai-org/GLM-4.5",
# Qwen models
"Qwen/Qwen3-235B-A22B-Instruct-2507",
"Qwen/Qwen3-Coder-480B-A35B-Instruct",
"Qwen/Qwen3-235B-A22B-Thinking-2507",
# moonshotai
"moonshotai/Kimi-K2-Instruct",
"moonshotai/Kimi-K2.5",
# MiniMaxAI
"MiniMaxAI/MiniMax-M2.5",
# meta models
"meta-llama/Llama-3.1-8B-Instruct",
"meta-llama/Llama-3.3-70B-Instruct",
"meta-llama/Llama-4-Scout-17B-16E-Instruct",
# deepseek-ai
"deepseek-ai/DeepSeek-V3.1",
"deepseek-ai/DeepSeek-R1-0528",
"deepseek-ai/DeepSeek-V3-0324",
# microsoft
"microsoft/Phi-4-mini-instruct",
]
)
modelscope_models: List = [
# LLM-Research models
"LLM-Research/c4ai-command-r-plus-08-2024",
"LLM-Research/Llama-4-Scout-17B-16E-Instruct",
"LLM-Research/Llama-4-Maverick-17B-128E-Instruct",
# Mistral models
"mistralai/Mistral-Small-Instruct-2409",
"mistralai/Ministral-8B-Instruct-2410",
"mistralai/Mistral-Large-Instruct-2407",
"Qwen/Qwen2.5-Coder-32B-Instruct",
"Qwen/Qwen2.5-Coder-14B-Instruct",
"Qwen/Qwen2.5-Coder-7B-Instruct",
"Qwen/Qwen2.5-72B-Instruct",
"Qwen/Qwen2.5-32B-Instruct",
"Qwen/Qwen2.5-14B-Instruct",
"Qwen/Qwen2.5-7B-Instruct",
"Qwen/QwQ-32B-Preview",
"opencompass/CompassJudger-1-32B-Instruct",
"Qwen/QVQ-72B-Preview",
"Qwen/Qwen2-VL-7B-Instruct",
"Qwen/Qwen2.5-14B-Instruct-1M",
"Qwen/Qwen2.5-7B-Instruct-1M",
"Qwen/Qwen2.5-VL-3B-Instruct",
"Qwen/Qwen2.5-VL-7B-Instruct",
"Qwen/Qwen2.5-VL-72B-Instruct",
"deepseek-ai/DeepSeek-V3",
"Qwen/QwQ-32B",
"XGenerationLab/XiYanSQL-QwenCoder-32B-2412",
"Qwen/Qwen2.5-VL-32B-Instruct",
"LLM-Research/Llama-4-Scout-17B-16E-Instruct",
"LLM-Research/Llama-4-Maverick-17B-128E-Instruct",
# Qwen3 series
"Qwen/Qwen3-0.6B",
"Qwen/Qwen3-1.7B",
"Qwen/Qwen3-4B",
@ -1120,14 +1132,36 @@ modelscope_models: List = [
"Qwen/Qwen3-30B-A3B",
"Qwen/Qwen3-32B",
"Qwen/Qwen3-235B-A22B",
"deepseek-ai/DeepSeek-R1-0528",
"MiniMax/MiniMax-M1-80k",
"Qwen/Qwen3-235B-A22B-Instruct-2507",
"Qwen/Qwen3-Coder-480B-A35B-Instruct",
"Qwen/Qwen3-235B-A22B-Thinking-2507",
"ZhipuAI/GLM-4.5",
"Qwen/Qwen3-30B-A3B-Thinking-2507",
"Qwen/Qwen3-Coder-30B-A3B-Instruct",
# Qwen3.5 series (new)
"Qwen/Qwen3.5-0.6B",
"Qwen/Qwen3.5-1.7B",
"Qwen/Qwen3.5-3B",
"Qwen/Qwen3.5-7B",
"Qwen/Qwen3.5-14B",
"Qwen/Qwen3.5-30B-A3B",
"Qwen/Qwen3.5-32B",
"Qwen/Qwen3.5-235B-A22B",
"Qwen/Qwen3.5-235B-A22B-Instruct-2507",
"Qwen/Qwen3.5-Coder-480B-A35B-Instruct",
"Qwen/Qwen3.5-235B-A22B-Thinking-2507",
"Qwen/Qwen3.5-30B-A3B-Thinking-2507",
"Qwen/Qwen3.5-Coder-30B-A3B-Instruct",
# Other models
"Qwen/QwQ-32B-Preview",
"opencompass/CompassJudger-1-32B-Instruct",
"Qwen/QVQ-72B-Preview",
"Qwen/Qwen2-VL-7B-Instruct",
"deepseek-ai/DeepSeek-V3",
"Qwen/QwQ-32B",
"XGenerationLab/XiYanSQL-QwenCoder-32B-2412",
"deepseek-ai/DeepSeek-R1-0528",
"MiniMax/MiniMax-M1-80k",
"ZhipuAI/GLM-4.5",
]
nebius_embedding_models: List = [

View file

@ -61,7 +61,7 @@ class TestModelScopeConfig:
# Set up environment variables for the test
api_key = "fake-modelscope-key"
api_base = "https://api-inference.modelscope.cn/v1"
model = "Qwen/Qwen3-8B"
model = "modelscope/Qwen/Qwen3-8B" # Use modelscope/ prefix to specify provider
model_name = "Qwen3-8B" # The actual model name without provider prefix
# Mock the HTTP request to the ModelScope API