fix: enable JSON-only provider routing + add xiaomi_mimo provider (#18291)

* fix: enable JSON-only provider routing + add xiaomi_mimo provider

* docs: add xiaomi_mimo provider documentation
This commit is contained in:
Farhan Aulianda 2025-12-22 13:35:22 +07:00 • committed by GitHub
parent b7fbf3cd7e
commit 63c3a6e228
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
6 changed files with 172 additions and 9 deletions

View file

@ -0,0 +1,137 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# Xiaomi MiMo
https://platform.xiaomimimo.com/#/docs
:::tip
**We support ALL Xiaomi MiMo models, just set `model=xiaomi_mimo/<any-model-on-xiaomi-mimo>` as a prefix when sending litellm requests**
:::
## API Key
```python
# env variable
os.environ['XIAOMI_MIMO_API_KEY']
```
## Sample Usage
```python
from litellm import completion
import os
os.environ['XIAOMI_MIMO_API_KEY'] = ""
response = completion(
model="xiaomi_mimo/mimo-v2-flash",
messages=[
{
"role": "user",
"content": "What's the weather like in Boston today in Fahrenheit?",
}
],
max_tokens=1024,
temperature=0.3,
top_p=0.95,
)
print(response)
```
## Sample Usage - Streaming
```python
from litellm import completion
import os
os.environ['XIAOMI_MIMO_API_KEY'] = ""
response = completion(
model="xiaomi_mimo/mimo-v2-flash",
messages=[
{
"role": "user",
"content": "What's the weather like in Boston today in Fahrenheit?",
}
],
stream=True,
max_tokens=1024,
temperature=0.3,
top_p=0.95,
)
for chunk in response:
print(chunk)
```
## Usage with LiteLLM Proxy Server
Here's how to call a Xiaomi MiMo model with the LiteLLM Proxy Server
1. Modify the config.yaml
```yaml
model_list:
- model_name: my-model
litellm_params:
model: xiaomi_mimo/<your-model-name> # add xiaomi_mimo/ prefix to route as Xiaomi MiMo provider
api_key: api-key # api key to send your model
```
2. Start the proxy
```bash
$ litellm --config /path/to/config.yaml
```
3. Send Request to LiteLLM Proxy Server
<Tabs>
<TabItem value="openai" label="OpenAI Python v1.0.0+">
```python
import openai
client = openai.OpenAI(
api_key="sk-1234", # pass litellm proxy key, if you're using virtual keys
base_url="http://0.0.0.0:4000" # litellm-proxy-base url
)
response = client.chat.completions.create(
model="my-model",
messages = [
{
"role": "user",
"content": "what llm are you"
}
],
)
print(response)
```
</TabItem>
<TabItem value="curl" label="curl">
```shell
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Authorization: Bearer sk-1234' \
--header 'Content-Type: application/json' \
--data '{
"model": "my-model",
"messages": [
{
"role": "user",
"content": "what llm are you"
}
],
}'
```
</TabItem>
</Tabs>
## Supported Models
| Model Name | Usage |
|------------|-------|
| mimo-v2-flash | `completion(model="xiaomi_mimo/mimo-v2-flash", messages)` |

View file

@ -289,7 +289,7 @@ const sidebars = {
label: "All Endpoints (Swagger)",
href: "https://litellm-api.up.railway.app/",
},
"proxy/enterprise",
"proxy/enterprise",
{
type: "category",
label: "Authentication",
@ -470,10 +470,10 @@ const sidebars = {
"proxy/managed_finetuning",
]
},
"generateContent",
"apply_guardrail",
"bedrock_invoke",
"interactions",
"generateContent",
"apply_guardrail",
"bedrock_invoke",
"interactions",
{
type: "category",
label: "/images",
@ -783,6 +783,7 @@ const sidebars = {
]
},
"providers/xai",
"providers/xiaomi_mimo",
"providers/xinference",
"providers/zai",
],

View file

@ -5,6 +5,7 @@ import httpx
import litellm
from litellm.constants import REPLICATE_MODEL_NAME_WITH_ID_LENGTH
from litellm.secret_managers.main import get_secret, get_secret_str
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
from ..types.router import LiteLLM_Params
@ -155,6 +156,17 @@ def get_llm_provider( # noqa: PLR0915
if api_key and api_key.startswith("os.environ/"):
dynamic_api_key = get_secret_str(api_key)
# Check JSON-configured providers FIRST (before enum-based provider_list)
provider_prefix = model.split("/", 1)[0]
if len(model.split("/")) > 1 and JSONProviderRegistry.exists(provider_prefix):
return _get_openai_compatible_provider_info(
model=model,
api_base=api_base,
api_key=api_key,
dynamic_api_key=dynamic_api_key,
)
# check if llm provider part of model name
if (

View file

@ -555,9 +555,13 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM):
provider_config: Optional[BaseConfig] = None
if custom_llm_provider is not None and model is not None:
provider_config = ProviderConfigManager.get_provider_chat_config(
model=model, provider=LlmProviders(custom_llm_provider)
)
try:
provider_config = ProviderConfigManager.get_provider_chat_config(
model=model, provider=LlmProviders(custom_llm_provider)
)
except ValueError:
# JSON-configured providers may not be in LlmProviders enum
provider_config = None
if provider_config is None:
provider_config = OpenAIConfig()

View file

@ -18,5 +18,12 @@
"veniceai": {
"base_url": "https://api.venice.ai/api/v1",
"api_key_env": "VENICE_AI_API_KEY"
},
"xiaomi_mimo": {
"base_url": "https://api.xiaomimimo.com/v1",
"api_key_env": "XIAOMI_MIMO_API_KEY",
"param_mappings": {
"max_completion_tokens": "max_tokens"
}
}
}
}

View file

@ -68,6 +68,7 @@ from litellm.constants import (
DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT,
)
from litellm.exceptions import LiteLLMUnknownProvider
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
from litellm.integrations.custom_logger import CustomLogger
from litellm.litellm_core_utils.asyncify import run_async_function
from litellm.litellm_core_utils.audio_utils.utils import (
@ -2263,6 +2264,7 @@ def completion( # type: ignore # noqa: PLR0915
or custom_llm_provider == "wandb"
or custom_llm_provider == "clarifai"
or custom_llm_provider in litellm.openai_compatible_providers
or JSONProviderRegistry.exists(custom_llm_provider) # JSON-configured providers
or "ft:gpt-3.5-turbo" in model # finetune gpt-3.5-turbo
): # allow user to make an openai call with a custom base
# note: if a user sets a custom base - we should ensure this works