diff --git a/litellm/__init__.py b/litellm/__init__.py index 405a83eea92..e0fb010c1d4 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -873,9 +873,9 @@ def add_known_models(model_cost_map: Optional[Dict] = None): elif value.get("litellm_provider") == "heroku": heroku_models.add(key) elif value.get("litellm_provider") == "dashscope": - dashscope_models.append(key) + dashscope_models.add(key) elif value.get("litellm_provider") == "modelscope": - modelscope_models.append(key) + modelscope_models.add(key) elif value.get("litellm_provider") == "moonshot": moonshot_models.add(key) elif value.get("litellm_provider") == "publicai": @@ -1239,30 +1239,14 @@ from .llms.topaz.common_utils import TopazModelInfo # OpenAIGPTConfig, OpenAIGPT5Config, etc. are lazy loaded - instances will be created on first access from .llms.xai.common_utils import XAIModelInfo -from .llms.azure.chat.gpt_transformation import AzureOpenAIConfig -from .llms.azure.completion.transformation import AzureOpenAITextConfig -from .llms.hosted_vllm.chat.transformation import HostedVLLMChatConfig -from .llms.llamafile.chat.transformation import LlamafileChatConfig -from .llms.litellm_proxy.chat.transformation import LiteLLMProxyChatConfig -from .llms.vllm.completion.transformation import VLLMConfig -from .llms.deepseek.chat.transformation import DeepSeekChatConfig -from .llms.lm_studio.chat.transformation import LMStudioChatConfig -from .llms.lm_studio.embed.transformation import LmStudioEmbeddingConfig -from .llms.nscale.chat.transformation import NscaleConfig -from .llms.perplexity.chat.transformation import PerplexityChatConfig -from .llms.azure.chat.o_series_transformation import AzureOpenAIO1Config -from .llms.watsonx.completion.transformation import IBMWatsonXAIConfig -from .llms.watsonx.chat.transformation import IBMWatsonXChatConfig -from .llms.watsonx.embed.transformation import IBMWatsonXEmbeddingConfig -from .llms.github_copilot.chat.transformation import GithubCopilotConfig -from .llms.nebius.chat.transformation import NebiusConfig -from .llms.dashscope.chat.transformation import DashScopeChatConfig -from .llms.modelscope.chat.transformation import ModelScopeChatConfig -from .llms.moonshot.chat.transformation import MoonshotChatConfig -from .llms.v0.chat.transformation import V0ChatConfig -from .llms.morph.chat.transformation import MorphChatConfig -from .llms.lambda_ai.chat.transformation import LambdaAIChatConfig -from .llms.hyperbolic.chat.transformation import HyperbolicChatConfig +# PublicAI now uses JSON-based configuration (see litellm/llms/openai_like/providers.json) +# All remaining configs are now lazy loaded - see _lazy_imports_registry.py + +# Import LlmProviders here (before main import) because it's imported during import time +# in multiple places including openai.py (via main import) +from litellm.types.utils import LlmProviders + +## Lazy loading this is not straightforward, will leave it here for now. from .main import * # type: ignore from .compression import compress # type: ignore[no-redef] @@ -1958,6 +1942,9 @@ if TYPE_CHECKING: from .llms.dashscope.rerank.transformation import ( DashScopeRerankConfig as DashScopeRerankConfig, ) + from .llms.modelscope.chat.transformation import ( + ModelScopeChatConfig as ModelScopeChatConfig, + ) from .llms.moonshot.chat.transformation import ( MoonshotChatConfig as MoonshotChatConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index bdc3289b87c..91e95b2d790 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -304,6 +304,7 @@ LLM_CONFIG_NAMES = ( "GigaChatConfig", "GigaChatEmbeddingConfig", "DashScopeChatConfig", + "ModelScopeChatConfig", "MoonshotChatConfig", "DockerModelRunnerChatConfig", "V0ChatConfig", @@ -1150,6 +1151,10 @@ _LLM_CONFIGS_IMPORT_MAP = { ".llms.dashscope.chat.transformation", "DashScopeChatConfig", ), + "ModelScopeChatConfig": ( + ".llms.modelscope.chat.transformation", + "ModelScopeChatConfig", + ), "MoonshotChatConfig": (".llms.moonshot.chat.transformation", "MoonshotChatConfig"), "DockerModelRunnerChatConfig": ( ".llms.docker_model_runner.chat.transformation", diff --git a/litellm/constants.py b/litellm/constants.py index 3a6a6c8a31b..3c439f9a36d 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1085,33 +1085,45 @@ dashscope_models: set = set( ] ) +WANDB_MODELS: set = set( + [ + # openai models + "openai/gpt-oss-120b", + "openai/gpt-oss-20b", + # zai-org models + "zai-org/GLM-4.5", + # Qwen models + "Qwen/Qwen3-235B-A22B-Instruct-2507", + "Qwen/Qwen3-Coder-480B-A35B-Instruct", + "Qwen/Qwen3-235B-A22B-Thinking-2507", + # moonshotai + "moonshotai/Kimi-K2-Instruct", + "moonshotai/Kimi-K2.5", + # MiniMaxAI + "MiniMaxAI/MiniMax-M2.5", + # meta models + "meta-llama/Llama-3.1-8B-Instruct", + "meta-llama/Llama-3.3-70B-Instruct", + "meta-llama/Llama-4-Scout-17B-16E-Instruct", + # deepseek-ai + "deepseek-ai/DeepSeek-V3.1", + "deepseek-ai/DeepSeek-R1-0528", + "deepseek-ai/DeepSeek-V3-0324", + # microsoft + "microsoft/Phi-4-mini-instruct", + ] +) + modelscope_models: List = [ + # LLM-Research models "LLM-Research/c4ai-command-r-plus-08-2024", + "LLM-Research/Llama-4-Scout-17B-16E-Instruct", + "LLM-Research/Llama-4-Maverick-17B-128E-Instruct", + # Mistral models "mistralai/Mistral-Small-Instruct-2409", "mistralai/Ministral-8B-Instruct-2410", "mistralai/Mistral-Large-Instruct-2407", - "Qwen/Qwen2.5-Coder-32B-Instruct", - "Qwen/Qwen2.5-Coder-14B-Instruct", - "Qwen/Qwen2.5-Coder-7B-Instruct", - "Qwen/Qwen2.5-72B-Instruct", - "Qwen/Qwen2.5-32B-Instruct", - "Qwen/Qwen2.5-14B-Instruct", - "Qwen/Qwen2.5-7B-Instruct", - "Qwen/QwQ-32B-Preview", - "opencompass/CompassJudger-1-32B-Instruct", - "Qwen/QVQ-72B-Preview", - "Qwen/Qwen2-VL-7B-Instruct", - "Qwen/Qwen2.5-14B-Instruct-1M", - "Qwen/Qwen2.5-7B-Instruct-1M", - "Qwen/Qwen2.5-VL-3B-Instruct", - "Qwen/Qwen2.5-VL-7B-Instruct", - "Qwen/Qwen2.5-VL-72B-Instruct", - "deepseek-ai/DeepSeek-V3", - "Qwen/QwQ-32B", - "XGenerationLab/XiYanSQL-QwenCoder-32B-2412", - "Qwen/Qwen2.5-VL-32B-Instruct", - "LLM-Research/Llama-4-Scout-17B-16E-Instruct", - "LLM-Research/Llama-4-Maverick-17B-128E-Instruct", + # Qwen3 series "Qwen/Qwen3-0.6B", "Qwen/Qwen3-1.7B", "Qwen/Qwen3-4B", @@ -1120,14 +1132,36 @@ modelscope_models: List = [ "Qwen/Qwen3-30B-A3B", "Qwen/Qwen3-32B", "Qwen/Qwen3-235B-A22B", - "deepseek-ai/DeepSeek-R1-0528", - "MiniMax/MiniMax-M1-80k", "Qwen/Qwen3-235B-A22B-Instruct-2507", "Qwen/Qwen3-Coder-480B-A35B-Instruct", "Qwen/Qwen3-235B-A22B-Thinking-2507", - "ZhipuAI/GLM-4.5", "Qwen/Qwen3-30B-A3B-Thinking-2507", "Qwen/Qwen3-Coder-30B-A3B-Instruct", + # Qwen3.5 series (new) + "Qwen/Qwen3.5-0.6B", + "Qwen/Qwen3.5-1.7B", + "Qwen/Qwen3.5-3B", + "Qwen/Qwen3.5-7B", + "Qwen/Qwen3.5-14B", + "Qwen/Qwen3.5-30B-A3B", + "Qwen/Qwen3.5-32B", + "Qwen/Qwen3.5-235B-A22B", + "Qwen/Qwen3.5-235B-A22B-Instruct-2507", + "Qwen/Qwen3.5-Coder-480B-A35B-Instruct", + "Qwen/Qwen3.5-235B-A22B-Thinking-2507", + "Qwen/Qwen3.5-30B-A3B-Thinking-2507", + "Qwen/Qwen3.5-Coder-30B-A3B-Instruct", + # Other models + "Qwen/QwQ-32B-Preview", + "opencompass/CompassJudger-1-32B-Instruct", + "Qwen/QVQ-72B-Preview", + "Qwen/Qwen2-VL-7B-Instruct", + "deepseek-ai/DeepSeek-V3", + "Qwen/QwQ-32B", + "XGenerationLab/XiYanSQL-QwenCoder-32B-2412", + "deepseek-ai/DeepSeek-R1-0528", + "MiniMax/MiniMax-M1-80k", + "ZhipuAI/GLM-4.5", ] nebius_embedding_models: List = [ diff --git a/tests/test_litellm/llms/modelscope/test_modelscope_chat_transformation.py b/tests/test_litellm/llms/modelscope/test_modelscope_chat_transformation.py index 7e233922189..c868eaaacaa 100644 --- a/tests/test_litellm/llms/modelscope/test_modelscope_chat_transformation.py +++ b/tests/test_litellm/llms/modelscope/test_modelscope_chat_transformation.py @@ -61,7 +61,7 @@ class TestModelScopeConfig: # Set up environment variables for the test api_key = "fake-modelscope-key" api_base = "https://api-inference.modelscope.cn/v1" - model = "Qwen/Qwen3-8B" + model = "modelscope/Qwen/Qwen3-8B" # Use modelscope/ prefix to specify provider model_name = "Qwen3-8B" # The actual model name without provider prefix # Mock the HTTP request to the ModelScope API