feat(sail): add Sail as an OpenAI-compatible provider

This commit is contained in:
shrey kharbanda 2026-09-24 00:45:46 +00:00
parent 0c1c3e18d5
commit c6e5d0fadd
10 changed files with 716 additions and 38 deletions

View file

@ -362,6 +362,7 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse
| [Recraft (`recraft`)](https://docs.litellm.ai/docs/providers/recraft) | | | | | ✅ | | | | | |
| [Replicate (`replicate`)](https://docs.litellm.ai/docs/providers/replicate) | ✅ | ✅ | ✅ | | | | | | | |
| [Sagemaker Chat (`sagemaker_chat`)](https://docs.litellm.ai/docs/providers/aws_sagemaker) | ✅ | ✅ | ✅ | | | | | | | |
| [Sail (`sail`)](https://docs.litellm.ai/docs/providers/sail) | ✅ | ✅ | ✅ | | | | | | | |
| [Sambanova (`sambanova`)](https://docs.litellm.ai/docs/providers/sambanova) | ✅ | ✅ | ✅ | | | | | | | |
| [Snowflake (`snowflake`)](https://docs.litellm.ai/docs/providers/snowflake) | ✅ | ✅ | ✅ | | | | | | | |
| [Text Completion Codestral (`text-completion-codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |

View file

@ -937,6 +937,7 @@ openai_compatible_endpoints: Final[list] = [
"https://api.meta.ai/v1",
"https://api.cognition.ai/v1",
"https://api.scx.ai/v1",
"https://api.sailresearch.com/v1",
"https://gigachat.devices.sberbank.ru/api/v1",
]
@ -1010,6 +1011,7 @@ openai_compatible_providers: Final[list] = [
"meta", # Meta Model API (Muse Spark) - JSON-configured provider
"cognition",
"scx-ai",
"sail",
]
OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers))

View file

@ -200,5 +200,14 @@
"temperature_max": 1.99
},
"supported_endpoints": ["/v1/chat/completions"]
},
"sail": {
"base_url": "https://api.sailresearch.com/v1",
"api_key_env": "SAIL_API_KEY",
"api_base_env": "SAIL_API_BASE",
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"param_mappings": {
"max_tokens": "max_completion_tokens"
}
}
}

View file

@ -57945,31 +57945,202 @@
"supports_reasoning": false,
"source": "https://pinstripes.io/"
},
"pinstripes/ps/deepseek-v4-flash": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"input_cost_per_token": 1e-07,
"output_cost_per_token": 2e-07,
"litellm_provider": "pinstripes",
"sail/moonshotai/Kimi-K3": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 2.5e-06,
"output_cost_per_token": 1.25e-05,
"cache_read_input_token_cost": 2.5e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_assistant_prefill": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://pinstripes.io/"
"source": "https://docs.sailresearch.com/models"
},
"pinstripes/ps/minimax-m2.7": {
"max_tokens": 1000192,
"max_input_tokens": 1000192,
"max_output_tokens": 1000192,
"input_cost_per_token": 2.55e-07,
"output_cost_per_token": 5.5e-07,
"litellm_provider": "pinstripes",
"sail/zai-org/GLM-5.3": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 9.8e-07,
"output_cost_per_token": 3.08e-06,
"cache_read_input_token_cost": 1.8e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_assistant_prefill": true,
"supports_reasoning": false,
"source": "https://pinstripes.io/"
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/zai-org/GLM-5.3-Flash": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 1.1e-07,
"output_cost_per_token": 3.5e-07,
"cache_read_input_token_cost": 2e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/deepseek-ai/DeepSeek-V4-Pro-0813": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 9.2e-07,
"output_cost_per_token": 2.77e-06,
"cache_read_input_token_cost": 4e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/deepseek-ai/DeepSeek-V4-Flash-0731": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 9e-08,
"output_cost_per_token": 1.8e-07,
"cache_read_input_token_cost": 2e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/deepseek-ai/DeepSeek-V4.1-Flash": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 6e-07,
"cache_read_input_token_cost": 6e-09,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/moonshotai/Kimi-K2.6": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 1e-06,
"output_cost_per_token": 4e-06,
"cache_read_input_token_cost": 2e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/google/gemma-4-31B-it": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 6e-07,
"cache_read_input_token_cost": 2e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/nvidia/Gemma-4-31B-IT-NVFP4": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 4e-07,
"cache_read_input_token_cost": 7e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/google/gemma-4-12B-it": {
"max_tokens": 16384,
"max_input_tokens": 16384,
"max_output_tokens": 16384,
"input_cost_per_token": 3e-07,
"output_cost_per_token": 2e-06,
"cache_read_input_token_cost": 1.5e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/openai/gpt-oss-120b": {
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 6e-08,
"output_cost_per_token": 4e-07,
"cache_read_input_token_cost": 3e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/Qwen/Qwen3.6-35B-A3B": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
"cache_read_input_token_cost": 2e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"darkbloom/gemma-4-26b": {
"input_cost_per_token": 3e-08,

View file

@ -2029,6 +2029,23 @@
"search": true
}
},
"sail": {
"display_name": "Sail (`sail`)",
"url": "https://docs.litellm.ai/docs/providers/sail",
"endpoints": {
"chat_completions": true,
"messages": false,
"responses": true,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_speech": false,
"moderations": false,
"batches": false,
"rerank": false,
"a2a": false
}
},
"sambanova": {
"display_name": "Sambanova (`sambanova`)",
"url": "https://docs.litellm.ai/docs/providers/sambanova",

View file

@ -2949,6 +2949,34 @@
],
"default_model_placeholder": "gpt-3.5-turbo"
},
{
"provider": "Sail",
"provider_display_name": "Sail",
"litellm_provider": "sail",
"credential_fields": [
{
"key": "api_base",
"label": "API Base",
"placeholder": "https://api.sailresearch.com/v1",
"tooltip": null,
"required": false,
"field_type": "text",
"options": null,
"default_value": null
},
{
"key": "api_key",
"label": "API Key",
"placeholder": null,
"tooltip": null,
"required": true,
"field_type": "password",
"options": null,
"default_value": null
}
],
"default_model_placeholder": "sail/zai-org/GLM-5.3"
},
{
"provider": "Sambanova",
"provider_display_name": "Sambanova",

View file

@ -4297,6 +4297,7 @@ class LlmProviders(str, Enum):
PINSTRIPES = "pinstripes"
COGNITION = "cognition"
SCX_AI = "scx-ai"
SAIL = "sail"
DARKBLOOM = "darkbloom"
META = "meta"
LITELLM_AGENT = "litellm_agent"

View file

@ -57945,31 +57945,202 @@
"supports_reasoning": false,
"source": "https://pinstripes.io/"
},
"pinstripes/ps/deepseek-v4-flash": {
"max_tokens": 163840,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"input_cost_per_token": 1e-07,
"output_cost_per_token": 2e-07,
"litellm_provider": "pinstripes",
"sail/moonshotai/Kimi-K3": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 2.5e-06,
"output_cost_per_token": 1.25e-05,
"cache_read_input_token_cost": 2.5e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_assistant_prefill": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://pinstripes.io/"
"source": "https://docs.sailresearch.com/models"
},
"pinstripes/ps/minimax-m2.7": {
"max_tokens": 1000192,
"max_input_tokens": 1000192,
"max_output_tokens": 1000192,
"input_cost_per_token": 2.55e-07,
"output_cost_per_token": 5.5e-07,
"litellm_provider": "pinstripes",
"sail/zai-org/GLM-5.3": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 9.8e-07,
"output_cost_per_token": 3.08e-06,
"cache_read_input_token_cost": 1.8e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_assistant_prefill": true,
"supports_reasoning": false,
"source": "https://pinstripes.io/"
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/zai-org/GLM-5.3-Flash": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 1.1e-07,
"output_cost_per_token": 3.5e-07,
"cache_read_input_token_cost": 2e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/deepseek-ai/DeepSeek-V4-Pro-0813": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 9.2e-07,
"output_cost_per_token": 2.77e-06,
"cache_read_input_token_cost": 4e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/deepseek-ai/DeepSeek-V4-Flash-0731": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 9e-08,
"output_cost_per_token": 1.8e-07,
"cache_read_input_token_cost": 2e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/deepseek-ai/DeepSeek-V4.1-Flash": {
"max_tokens": 1048576,
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"input_cost_per_token": 1.5e-07,
"output_cost_per_token": 6e-07,
"cache_read_input_token_cost": 6e-09,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/moonshotai/Kimi-K2.6": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 1e-06,
"output_cost_per_token": 4e-06,
"cache_read_input_token_cost": 2e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/google/gemma-4-31B-it": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 6e-07,
"cache_read_input_token_cost": 2e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/nvidia/Gemma-4-31B-IT-NVFP4": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 4e-07,
"cache_read_input_token_cost": 7e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/google/gemma-4-12B-it": {
"max_tokens": 16384,
"max_input_tokens": 16384,
"max_output_tokens": 16384,
"input_cost_per_token": 3e-07,
"output_cost_per_token": 2e-06,
"cache_read_input_token_cost": 1.5e-07,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/openai/gpt-oss-120b": {
"max_tokens": 131072,
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"input_cost_per_token": 6e-08,
"output_cost_per_token": 4e-07,
"cache_read_input_token_cost": 3e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"source": "https://docs.sailresearch.com/models"
},
"sail/Qwen/Qwen3.6-35B-A3B": {
"max_tokens": 262144,
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
"cache_read_input_token_cost": 2e-08,
"litellm_provider": "sail",
"mode": "chat",
"supports_function_calling": true,
"supports_tool_choice": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_reasoning": true,
"supports_vision": true,
"source": "https://docs.sailresearch.com/models"
},
"darkbloom/gemma-4-26b": {
"input_cost_per_token": 3e-08,

View file

@ -2263,6 +2263,23 @@
"search": true
}
},
"sail": {
"display_name": "Sail (`sail`)",
"url": "https://docs.litellm.ai/docs/providers/sail",
"endpoints": {
"chat_completions": true,
"messages": false,
"responses": true,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_speech": false,
"moderations": false,
"batches": false,
"rerank": false,
"a2a": false
}
},
"sambanova": {
"display_name": "Sambanova (`sambanova`)",
"url": "https://docs.litellm.ai/docs/providers/sambanova",

View file

@ -0,0 +1,261 @@
"""
Tests for the Sail (sailresearch.com) JSON-configured provider.
Each test asserts the outbound HTTP request that litellm would send to Sail,
via a mocked httpx transport, rather than asserting registry contents.
"""
import json
import httpx
import pytest
import respx
import litellm
SAIL_BASE_URL = "https://api.sailresearch.com/v1"
SAIL_CHAT_COMPLETIONS = f"{SAIL_BASE_URL}/chat/completions"
SAIL_RESPONSES = f"{SAIL_BASE_URL}/responses"
MODEL = "sail/zai-org/GLM-5.3"
def _chat_completion_payload() -> dict:
return {
"id": "chatcmpl-sail",
"object": "chat.completion",
"created": 1234567890,
"model": "zai-org/GLM-5.3",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "sail response"},
"finish_reason": "stop",
}
],
"usage": {"prompt_tokens": 2, "completion_tokens": 2, "total_tokens": 4},
}
def _chat_completion_stream() -> str:
chunks = [
{
"id": "chatcmpl-sail-stream",
"object": "chat.completion.chunk",
"created": 1234567890,
"model": "zai-org/GLM-5.3",
"choices": [
{
"index": 0,
"delta": {"role": "assistant", "content": "sail"},
"finish_reason": None,
}
],
},
{
"id": "chatcmpl-sail-stream",
"object": "chat.completion.chunk",
"created": 1234567890,
"model": "zai-org/GLM-5.3",
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
},
]
return "".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks) + "data: [DONE]\n\n"
def _responses_payload() -> dict:
return {
"id": "resp_sail",
"object": "response",
"created_at": 1234567890,
"status": "completed",
"model": "zai-org/GLM-5.3",
"output": [],
"parallel_tool_calls": True,
"usage": {"input_tokens": 2, "output_tokens": 2, "total_tokens": 4},
"error": None,
}
@pytest.fixture(autouse=True)
def _sail_env(monkeypatch: pytest.MonkeyPatch):
litellm.disable_aiohttp_transport = True
monkeypatch.setenv("SAIL_API_KEY", "sk-sail-test")
monkeypatch.delenv("SAIL_API_BASE", raising=False)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
class TestSailRequestShape:
@pytest.mark.respx()
def test_sail_chat_completions_url_and_auth(self, respx_mock: respx.Router):
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
litellm.completion(
model=MODEL,
messages=[{"role": "user", "content": "hi"}],
)
assert len(respx_mock.calls) == 1
request = respx_mock.calls[0].request
assert request.url == SAIL_CHAT_COMPLETIONS
assert request.headers["Authorization"] == "Bearer sk-sail-test"
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
_, provider, _, _ = get_llm_provider(
model=MODEL, custom_llm_provider=None, api_base=None, api_key=None
)
assert provider == "sail"
@pytest.mark.respx()
def test_sail_api_base_env_overrides_url(
self, respx_mock: respx.Router, monkeypatch: pytest.MonkeyPatch
):
monkeypatch.setenv("SAIL_API_BASE", "https://sail.internal.example/v2")
route = respx_mock.post("https://sail.internal.example/v2/chat/completions").respond(
json=_chat_completion_payload()
)
litellm.completion(model=MODEL, messages=[{"role": "user", "content": "hi"}])
assert route.called
@pytest.mark.respx()
def test_sail_explicit_api_base_trailing_slash_no_double_slash(self, respx_mock: respx.Router):
route = respx_mock.post("https://custom.sail.example/v1/chat/completions").respond(
json=_chat_completion_payload()
)
litellm.completion(
model=MODEL,
messages=[{"role": "user", "content": "hi"}],
api_base="https://custom.sail.example/v1/",
)
assert route.called
@pytest.mark.respx()
def test_sail_max_tokens_sent_as_max_completion_tokens(self, respx_mock: respx.Router):
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
litellm.completion(
model=MODEL,
messages=[{"role": "user", "content": "hi"}],
max_tokens=100,
)
body = json.loads(respx_mock.calls[0].request.content)
assert body["max_completion_tokens"] == 100
assert "max_tokens" not in body
@pytest.mark.respx()
def test_sail_no_metadata_key_when_caller_passes_none(self, respx_mock: respx.Router):
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
litellm.completion(model=MODEL, messages=[{"role": "user", "content": "hi"}])
body = json.loads(respx_mock.calls[0].request.content)
assert "metadata" not in body
@pytest.mark.respx()
@pytest.mark.parametrize("stream", [False, True], ids=["non-streaming", "streaming"])
def test_sail_completion_window_via_extra_body(self, respx_mock: respx.Router, stream: bool):
if stream:
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(
content=_chat_completion_stream(),
headers={"content-type": "text/event-stream"},
)
else:
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
response = litellm.completion(
model=MODEL,
messages=[{"role": "user", "content": "hi"}],
stream=stream,
extra_body={"metadata": {"completion_window": "balanced"}},
)
if stream:
list(response)
body = json.loads(respx_mock.calls[0].request.content)
assert body["metadata"] == {"completion_window": "balanced"}
@pytest.mark.respx()
def test_sail_tools_survive_to_request_body(self, respx_mock: respx.Router):
respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload())
litellm.completion(
model=MODEL,
messages=[{"role": "user", "content": "what is the weather"}],
tools=[
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get the weather for a city",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
},
}
],
)
body = json.loads(respx_mock.calls[0].request.content)
assert body["tools"][0]["function"]["name"] == "get_weather"
@pytest.mark.asyncio
@pytest.mark.respx()
async def test_sail_aresponses_posts_metadata_and_background(self, respx_mock: respx.Router):
respx_mock.post(SAIL_RESPONSES).respond(json=_responses_payload())
await litellm.aresponses(
model=MODEL,
input="hi",
metadata={"completion_window": "flex"},
background=True,
)
assert len(respx_mock.calls) == 1
request = respx_mock.calls[0].request
assert request.url == SAIL_RESPONSES
body = json.loads(request.content)
assert body["metadata"] == {"completion_window": "flex"}
assert body["background"] is True
class TestSailCostTracking:
def test_cached_tokens_billed_at_sail_cache_read_rate(self, monkeypatch: pytest.MonkeyPatch):
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
rates = litellm.model_cost[MODEL]
prompt_tokens = 1000
cached_tokens = 600
completion_tokens = 200
response = litellm.ModelResponse(
model="zai-org/GLM-5.3",
choices=[{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}],
usage=Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=prompt_tokens + completion_tokens,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens),
),
)
cost = litellm.completion_cost(
completion_response=response,
model=MODEL,
custom_llm_provider="sail",
)
expected = (
(prompt_tokens - cached_tokens) * rates["input_cost_per_token"]
+ cached_tokens * rates["cache_read_input_token_cost"]
+ completion_tokens * rates["output_cost_per_token"]
)
assert cost == pytest.approx(expected)