mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
feat(sail): add Sail as an OpenAI-compatible provider
This commit is contained in:
parent
0c1c3e18d5
commit
c6e5d0fadd
10 changed files with 716 additions and 38 deletions
|
|
@ -362,6 +362,7 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse
|
|||
| [Recraft (`recraft`)](https://docs.litellm.ai/docs/providers/recraft) | | | | | ✅ | | | | | |
|
||||
| [Replicate (`replicate`)](https://docs.litellm.ai/docs/providers/replicate) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Sagemaker Chat (`sagemaker_chat`)](https://docs.litellm.ai/docs/providers/aws_sagemaker) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Sail (`sail`)](https://docs.litellm.ai/docs/providers/sail) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Sambanova (`sambanova`)](https://docs.litellm.ai/docs/providers/sambanova) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Snowflake (`snowflake`)](https://docs.litellm.ai/docs/providers/snowflake) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Text Completion Codestral (`text-completion-codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
|
|
|
|||
|
|
@ -937,6 +937,7 @@ openai_compatible_endpoints: Final[list] = [
|
|||
"https://api.meta.ai/v1",
|
||||
"https://api.cognition.ai/v1",
|
||||
"https://api.scx.ai/v1",
|
||||
"https://api.sailresearch.com/v1",
|
||||
"https://gigachat.devices.sberbank.ru/api/v1",
|
||||
]
|
||||
|
||||
|
|
@ -1010,6 +1011,7 @@ openai_compatible_providers: Final[list] = [
|
|||
"meta", # Meta Model API (Muse Spark) - JSON-configured provider
|
||||
"cognition",
|
||||
"scx-ai",
|
||||
"sail",
|
||||
]
|
||||
|
||||
OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers))
|
||||
|
|
|
|||
|
|
@ -200,5 +200,14 @@
|
|||
"temperature_max": 1.99
|
||||
},
|
||||
"supported_endpoints": ["/v1/chat/completions"]
|
||||
},
|
||||
"sail": {
|
||||
"base_url": "https://api.sailresearch.com/v1",
|
||||
"api_key_env": "SAIL_API_KEY",
|
||||
"api_base_env": "SAIL_API_BASE",
|
||||
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
|
||||
"param_mappings": {
|
||||
"max_tokens": "max_completion_tokens"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -57945,31 +57945,202 @@
|
|||
"supports_reasoning": false,
|
||||
"source": "https://pinstripes.io/"
|
||||
},
|
||||
"pinstripes/ps/deepseek-v4-flash": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"litellm_provider": "pinstripes",
|
||||
"sail/moonshotai/Kimi-K3": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"output_cost_per_token": 1.25e-05,
|
||||
"cache_read_input_token_cost": 2.5e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://pinstripes.io/"
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"pinstripes/ps/minimax-m2.7": {
|
||||
"max_tokens": 1000192,
|
||||
"max_input_tokens": 1000192,
|
||||
"max_output_tokens": 1000192,
|
||||
"input_cost_per_token": 2.55e-07,
|
||||
"output_cost_per_token": 5.5e-07,
|
||||
"litellm_provider": "pinstripes",
|
||||
"sail/zai-org/GLM-5.3": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 9.8e-07,
|
||||
"output_cost_per_token": 3.08e-06,
|
||||
"cache_read_input_token_cost": 1.8e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_reasoning": false,
|
||||
"source": "https://pinstripes.io/"
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/zai-org/GLM-5.3-Flash": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 1.1e-07,
|
||||
"output_cost_per_token": 3.5e-07,
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/deepseek-ai/DeepSeek-V4-Pro-0813": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 9.2e-07,
|
||||
"output_cost_per_token": 2.77e-06,
|
||||
"cache_read_input_token_cost": 4e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/deepseek-ai/DeepSeek-V4-Flash-0731": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 9e-08,
|
||||
"output_cost_per_token": 1.8e-07,
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/deepseek-ai/DeepSeek-V4.1-Flash": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"output_cost_per_token": 6e-07,
|
||||
"cache_read_input_token_cost": 6e-09,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/moonshotai/Kimi-K2.6": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 4e-06,
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/google/gemma-4-31B-it": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 6e-07,
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/nvidia/Gemma-4-31B-IT-NVFP4": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"cache_read_input_token_cost": 7e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/google/gemma-4-12B-it": {
|
||||
"max_tokens": 16384,
|
||||
"max_input_tokens": 16384,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 2e-06,
|
||||
"cache_read_input_token_cost": 1.5e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/openai/gpt-oss-120b": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 6e-08,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/Qwen/Qwen3.6-35B-A3B": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"darkbloom/gemma-4-26b": {
|
||||
"input_cost_per_token": 3e-08,
|
||||
|
|
|
|||
|
|
@ -2029,6 +2029,23 @@
|
|||
"search": true
|
||||
}
|
||||
},
|
||||
"sail": {
|
||||
"display_name": "Sail (`sail`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/sail",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": false,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
"audio_transcriptions": false,
|
||||
"audio_speech": false,
|
||||
"moderations": false,
|
||||
"batches": false,
|
||||
"rerank": false,
|
||||
"a2a": false
|
||||
}
|
||||
},
|
||||
"sambanova": {
|
||||
"display_name": "Sambanova (`sambanova`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/sambanova",
|
||||
|
|
|
|||
|
|
@ -2949,6 +2949,34 @@
|
|||
],
|
||||
"default_model_placeholder": "gpt-3.5-turbo"
|
||||
},
|
||||
{
|
||||
"provider": "Sail",
|
||||
"provider_display_name": "Sail",
|
||||
"litellm_provider": "sail",
|
||||
"credential_fields": [
|
||||
{
|
||||
"key": "api_base",
|
||||
"label": "API Base",
|
||||
"placeholder": "https://api.sailresearch.com/v1",
|
||||
"tooltip": null,
|
||||
"required": false,
|
||||
"field_type": "text",
|
||||
"options": null,
|
||||
"default_value": null
|
||||
},
|
||||
{
|
||||
"key": "api_key",
|
||||
"label": "API Key",
|
||||
"placeholder": null,
|
||||
"tooltip": null,
|
||||
"required": true,
|
||||
"field_type": "password",
|
||||
"options": null,
|
||||
"default_value": null
|
||||
}
|
||||
],
|
||||
"default_model_placeholder": "sail/zai-org/GLM-5.3"
|
||||
},
|
||||
{
|
||||
"provider": "Sambanova",
|
||||
"provider_display_name": "Sambanova",
|
||||
|
|
|
|||
|
|
@ -4297,6 +4297,7 @@ class LlmProviders(str, Enum):
|
|||
PINSTRIPES = "pinstripes"
|
||||
COGNITION = "cognition"
|
||||
SCX_AI = "scx-ai"
|
||||
SAIL = "sail"
|
||||
DARKBLOOM = "darkbloom"
|
||||
META = "meta"
|
||||
LITELLM_AGENT = "litellm_agent"
|
||||
|
|
|
|||
|
|
@ -57945,31 +57945,202 @@
|
|||
"supports_reasoning": false,
|
||||
"source": "https://pinstripes.io/"
|
||||
},
|
||||
"pinstripes/ps/deepseek-v4-flash": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"litellm_provider": "pinstripes",
|
||||
"sail/moonshotai/Kimi-K3": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"output_cost_per_token": 1.25e-05,
|
||||
"cache_read_input_token_cost": 2.5e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://pinstripes.io/"
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"pinstripes/ps/minimax-m2.7": {
|
||||
"max_tokens": 1000192,
|
||||
"max_input_tokens": 1000192,
|
||||
"max_output_tokens": 1000192,
|
||||
"input_cost_per_token": 2.55e-07,
|
||||
"output_cost_per_token": 5.5e-07,
|
||||
"litellm_provider": "pinstripes",
|
||||
"sail/zai-org/GLM-5.3": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 9.8e-07,
|
||||
"output_cost_per_token": 3.08e-06,
|
||||
"cache_read_input_token_cost": 1.8e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_reasoning": false,
|
||||
"source": "https://pinstripes.io/"
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/zai-org/GLM-5.3-Flash": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 1.1e-07,
|
||||
"output_cost_per_token": 3.5e-07,
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/deepseek-ai/DeepSeek-V4-Pro-0813": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 9.2e-07,
|
||||
"output_cost_per_token": 2.77e-06,
|
||||
"cache_read_input_token_cost": 4e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/deepseek-ai/DeepSeek-V4-Flash-0731": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 9e-08,
|
||||
"output_cost_per_token": 1.8e-07,
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/deepseek-ai/DeepSeek-V4.1-Flash": {
|
||||
"max_tokens": 1048576,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 1048576,
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"output_cost_per_token": 6e-07,
|
||||
"cache_read_input_token_cost": 6e-09,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/moonshotai/Kimi-K2.6": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 4e-06,
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/google/gemma-4-31B-it": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 6e-07,
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/nvidia/Gemma-4-31B-IT-NVFP4": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"cache_read_input_token_cost": 7e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/google/gemma-4-12B-it": {
|
||||
"max_tokens": 16384,
|
||||
"max_input_tokens": 16384,
|
||||
"max_output_tokens": 16384,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 2e-06,
|
||||
"cache_read_input_token_cost": 1.5e-07,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/openai/gpt-oss-120b": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 6e-08,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"sail/Qwen/Qwen3.6-35B-A3B": {
|
||||
"max_tokens": 262144,
|
||||
"max_input_tokens": 262144,
|
||||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"litellm_provider": "sail",
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.sailresearch.com/models"
|
||||
},
|
||||
"darkbloom/gemma-4-26b": {
|
||||
"input_cost_per_token": 3e-08,
|
||||
|
|
|
|||
|
|
@ -2263,6 +2263,23 @@
|
|||
"search": true
|
||||
}
|
||||
},
|
||||
"sail": {
|
||||
"display_name": "Sail (`sail`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/sail",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": false,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
"audio_transcriptions": false,
|
||||
"audio_speech": false,
|
||||
"moderations": false,
|
||||
"batches": false,
|
||||
"rerank": false,
|
||||
"a2a": false
|
||||
}
|
||||
},
|
||||
"sambanova": {
|
||||
"display_name": "Sambanova (`sambanova`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/sambanova",
|
||||
|
|
|
|||
261
tests/test_litellm/llms/openai_like/test_sail_provider.py
Normal file
261
tests/test_litellm/llms/openai_like/test_sail_provider.py
Normal file
|
|
@ -0,0 +1,261 @@
|
|||
"""
|
||||
Tests for the Sail (sailresearch.com) JSON-configured provider.
|
||||
|
||||
Each test asserts the outbound HTTP request that litellm would send to Sail,
|
||||
via a mocked httpx transport, rather than asserting registry contents.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
import respx
|
||||
|
||||
import litellm
|
||||
|
||||
SAIL_BASE_URL = "https://api.sailresearch.com/v1"
|
||||
SAIL_CHAT_COMPLETIONS = f"{SAIL_BASE_URL}/chat/completions"
|
||||
SAIL_RESPONSES = f"{SAIL_BASE_URL}/responses"
|
||||
|
||||
MODEL = "sail/zai-org/GLM-5.3"
|
||||
|
||||
|
||||
def _chat_completion_payload() -> dict:
|
||||
return {
|
||||
"id": "chatcmpl-sail",
|
||||
"object": "chat.completion",
|
||||
"created": 1234567890,
|
||||
"model": "zai-org/GLM-5.3",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "sail response"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
"usage": {"prompt_tokens": 2, "completion_tokens": 2, "total_tokens": 4},
|
||||
}
|
||||
|
||||
|
||||
def _chat_completion_stream() -> str:
|
||||
chunks = [
|
||||
{
|
||||
"id": "chatcmpl-sail-stream",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1234567890,
|
||||
"model": "zai-org/GLM-5.3",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"role": "assistant", "content": "sail"},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"id": "chatcmpl-sail-stream",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1234567890,
|
||||
"model": "zai-org/GLM-5.3",
|
||||
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
|
||||
},
|
||||
]
|
||||
return "".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks) + "data: [DONE]\n\n"
|
||||
|
||||
|
||||
def _responses_payload() -> dict:
|
||||
return {
|
||||
"id": "resp_sail",
|
||||
"object": "response",
|
||||
"created_at": 1234567890,
|
||||
"status": "completed",
|
||||
"model": "zai-org/GLM-5.3",
|
||||
"output": [],
|
||||
"parallel_tool_calls": True,
|
||||
"usage": {"input_tokens": 2, "output_tokens": 2, "total_tokens": 4},
|
||||
"error": None,
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _sail_env(monkeypatch: pytest.MonkeyPatch):
|
||||
litellm.disable_aiohttp_transport = True
|
||||
monkeypatch.setenv("SAIL_API_KEY", "sk-sail-test")
|
||||
monkeypatch.delenv("SAIL_API_BASE", raising=False)
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
|
||||
class TestSailRequestShape:
|
||||
@pytest.mark.respx()
|
||||
def test_sail_chat_completions_url_and_auth(self, respx_mock: respx.Router):
|
||||
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
|
||||
|
||||
litellm.completion(
|
||||
model=MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
|
||||
assert len(respx_mock.calls) == 1
|
||||
request = respx_mock.calls[0].request
|
||||
assert request.url == SAIL_CHAT_COMPLETIONS
|
||||
assert request.headers["Authorization"] == "Bearer sk-sail-test"
|
||||
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
|
||||
_, provider, _, _ = get_llm_provider(
|
||||
model=MODEL, custom_llm_provider=None, api_base=None, api_key=None
|
||||
)
|
||||
assert provider == "sail"
|
||||
|
||||
@pytest.mark.respx()
|
||||
def test_sail_api_base_env_overrides_url(
|
||||
self, respx_mock: respx.Router, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
monkeypatch.setenv("SAIL_API_BASE", "https://sail.internal.example/v2")
|
||||
route = respx_mock.post("https://sail.internal.example/v2/chat/completions").respond(
|
||||
json=_chat_completion_payload()
|
||||
)
|
||||
|
||||
litellm.completion(model=MODEL, messages=[{"role": "user", "content": "hi"}])
|
||||
|
||||
assert route.called
|
||||
|
||||
@pytest.mark.respx()
|
||||
def test_sail_explicit_api_base_trailing_slash_no_double_slash(self, respx_mock: respx.Router):
|
||||
route = respx_mock.post("https://custom.sail.example/v1/chat/completions").respond(
|
||||
json=_chat_completion_payload()
|
||||
)
|
||||
|
||||
litellm.completion(
|
||||
model=MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
api_base="https://custom.sail.example/v1/",
|
||||
)
|
||||
|
||||
assert route.called
|
||||
|
||||
@pytest.mark.respx()
|
||||
def test_sail_max_tokens_sent_as_max_completion_tokens(self, respx_mock: respx.Router):
|
||||
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
|
||||
|
||||
litellm.completion(
|
||||
model=MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
max_tokens=100,
|
||||
)
|
||||
|
||||
body = json.loads(respx_mock.calls[0].request.content)
|
||||
assert body["max_completion_tokens"] == 100
|
||||
assert "max_tokens" not in body
|
||||
|
||||
@pytest.mark.respx()
|
||||
def test_sail_no_metadata_key_when_caller_passes_none(self, respx_mock: respx.Router):
|
||||
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
|
||||
|
||||
litellm.completion(model=MODEL, messages=[{"role": "user", "content": "hi"}])
|
||||
|
||||
body = json.loads(respx_mock.calls[0].request.content)
|
||||
assert "metadata" not in body
|
||||
|
||||
@pytest.mark.respx()
|
||||
@pytest.mark.parametrize("stream", [False, True], ids=["non-streaming", "streaming"])
|
||||
def test_sail_completion_window_via_extra_body(self, respx_mock: respx.Router, stream: bool):
|
||||
if stream:
|
||||
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(
|
||||
content=_chat_completion_stream(),
|
||||
headers={"content-type": "text/event-stream"},
|
||||
)
|
||||
else:
|
||||
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
|
||||
|
||||
response = litellm.completion(
|
||||
model=MODEL,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
stream=stream,
|
||||
extra_body={"metadata": {"completion_window": "balanced"}},
|
||||
)
|
||||
if stream:
|
||||
list(response)
|
||||
|
||||
body = json.loads(respx_mock.calls[0].request.content)
|
||||
assert body["metadata"] == {"completion_window": "balanced"}
|
||||
|
||||
@pytest.mark.respx()
|
||||
def test_sail_tools_survive_to_request_body(self, respx_mock: respx.Router):
|
||||
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
|
||||
|
||||
litellm.completion(
|
||||
model=MODEL,
|
||||
messages=[{"role": "user", "content": "what is the weather"}],
|
||||
tools=[
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get the weather for a city",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"city": {"type": "string"}},
|
||||
"required": ["city"],
|
||||
},
|
||||
},
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
body = json.loads(respx_mock.calls[0].request.content)
|
||||
assert body["tools"][0]["function"]["name"] == "get_weather"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.respx()
|
||||
async def test_sail_aresponses_posts_metadata_and_background(self, respx_mock: respx.Router):
|
||||
respx_mock.post(SAIL_RESPONSES).respond(json=_responses_payload())
|
||||
|
||||
await litellm.aresponses(
|
||||
model=MODEL,
|
||||
input="hi",
|
||||
metadata={"completion_window": "flex"},
|
||||
background=True,
|
||||
)
|
||||
|
||||
assert len(respx_mock.calls) == 1
|
||||
request = respx_mock.calls[0].request
|
||||
assert request.url == SAIL_RESPONSES
|
||||
body = json.loads(request.content)
|
||||
assert body["metadata"] == {"completion_window": "flex"}
|
||||
assert body["background"] is True
|
||||
|
||||
|
||||
class TestSailCostTracking:
|
||||
def test_cached_tokens_billed_at_sail_cache_read_rate(self, monkeypatch: pytest.MonkeyPatch):
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
rates = litellm.model_cost[MODEL]
|
||||
prompt_tokens = 1000
|
||||
cached_tokens = 600
|
||||
completion_tokens = 200
|
||||
|
||||
response = litellm.ModelResponse(
|
||||
model="zai-org/GLM-5.3",
|
||||
choices=[{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}],
|
||||
usage=Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens),
|
||||
),
|
||||
)
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model=MODEL,
|
||||
custom_llm_provider="sail",
|
||||
)
|
||||
|
||||
expected = (
|
||||
(prompt_tokens - cached_tokens) * rates["input_cost_per_token"]
|
||||
+ cached_tokens * rates["cache_read_input_token_cost"]
|
||||
+ completion_tokens * rates["output_cost_per_token"]
|
||||
)
|
||||
assert cost == pytest.approx(expected)
|
||||
Loading…
Add table
Reference in a new issue