diff --git a/README.md b/README.md index e927c80b8b4..98c5343daee 100644 --- a/README.md +++ b/README.md @@ -362,6 +362,7 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse | [Recraft (`recraft`)](https://docs.litellm.ai/docs/providers/recraft) | | | | | ✅ | | | | | | | [Replicate (`replicate`)](https://docs.litellm.ai/docs/providers/replicate) | ✅ | ✅ | ✅ | | | | | | | | | [Sagemaker Chat (`sagemaker_chat`)](https://docs.litellm.ai/docs/providers/aws_sagemaker) | ✅ | ✅ | ✅ | | | | | | | | +| [Sail (`sail`)](https://docs.litellm.ai/docs/providers/sail) | ✅ | ✅ | ✅ | | | | | | | | | [Sambanova (`sambanova`)](https://docs.litellm.ai/docs/providers/sambanova) | ✅ | ✅ | ✅ | | | | | | | | | [Snowflake (`snowflake`)](https://docs.litellm.ai/docs/providers/snowflake) | ✅ | ✅ | ✅ | | | | | | | | | [Text Completion Codestral (`text-completion-codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | | diff --git a/litellm/constants.py b/litellm/constants.py index e5b662bd515..5969820da14 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -937,6 +937,7 @@ openai_compatible_endpoints: Final[list] = [ "https://api.meta.ai/v1", "https://api.cognition.ai/v1", "https://api.scx.ai/v1", + "https://api.sailresearch.com/v1", "https://gigachat.devices.sberbank.ru/api/v1", ] @@ -1010,6 +1011,7 @@ openai_compatible_providers: Final[list] = [ "meta", # Meta Model API (Muse Spark) - JSON-configured provider "cognition", "scx-ai", + "sail", ] OPENAI_AUDIO_TRANSCRIPTION_PROVIDERS: Final = frozenset({"openai"} | frozenset(openai_compatible_providers)) diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index fe10293c420..194636da7ff 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -200,5 +200,14 @@ "temperature_max": 1.99 }, "supported_endpoints": ["/v1/chat/completions"] + }, + "sail": { + "base_url": "https://api.sailresearch.com/v1", + "api_key_env": "SAIL_API_KEY", + "api_base_env": "SAIL_API_BASE", + "supported_endpoints": ["/v1/chat/completions", "/v1/responses"], + "param_mappings": { + "max_tokens": "max_completion_tokens" + } } } diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ebea1a6044d..1fc08c1175f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -57945,31 +57945,202 @@ "supports_reasoning": false, "source": "https://pinstripes.io/" }, - "pinstripes/ps/deepseek-v4-flash": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 1e-07, - "output_cost_per_token": 2e-07, - "litellm_provider": "pinstripes", + "sail/moonshotai/Kimi-K3": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 2.5e-06, + "output_cost_per_token": 1.25e-05, + "cache_read_input_token_cost": 2.5e-07, + "litellm_provider": "sail", "mode": "chat", "supports_function_calling": true, - "supports_assistant_prefill": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, "supports_reasoning": true, - "source": "https://pinstripes.io/" + "source": "https://docs.sailresearch.com/models" }, - "pinstripes/ps/minimax-m2.7": { - "max_tokens": 1000192, - "max_input_tokens": 1000192, - "max_output_tokens": 1000192, - "input_cost_per_token": 2.55e-07, - "output_cost_per_token": 5.5e-07, - "litellm_provider": "pinstripes", + "sail/zai-org/GLM-5.3": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 9.8e-07, + "output_cost_per_token": 3.08e-06, + "cache_read_input_token_cost": 1.8e-07, + "litellm_provider": "sail", "mode": "chat", "supports_function_calling": true, - "supports_assistant_prefill": true, - "supports_reasoning": false, - "source": "https://pinstripes.io/" + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/zai-org/GLM-5.3-Flash": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 1.1e-07, + "output_cost_per_token": 3.5e-07, + "cache_read_input_token_cost": 2e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/deepseek-ai/DeepSeek-V4-Pro-0813": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 9.2e-07, + "output_cost_per_token": 2.77e-06, + "cache_read_input_token_cost": 4e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/deepseek-ai/DeepSeek-V4-Flash-0731": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 1.8e-07, + "cache_read_input_token_cost": 2e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/deepseek-ai/DeepSeek-V4.1-Flash": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 6e-09, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/moonshotai/Kimi-K2.6": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 2e-07, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/google/gemma-4-31B-it": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 4e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 2e-07, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/nvidia/Gemma-4-31B-IT-NVFP4": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1.4e-07, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 7e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/google/gemma-4-12B-it": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 1.5e-07, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/openai/gpt-oss-120b": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 3e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/Qwen/Qwen3.6-35B-A3B": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 2e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" }, "darkbloom/gemma-4-26b": { "input_cost_per_token": 3e-08, diff --git a/litellm/provider_endpoints_support_backup.json b/litellm/provider_endpoints_support_backup.json index 1fcb7600a5e..eb3b380caca 100644 --- a/litellm/provider_endpoints_support_backup.json +++ b/litellm/provider_endpoints_support_backup.json @@ -2029,6 +2029,23 @@ "search": true } }, + "sail": { + "display_name": "Sail (`sail`)", + "url": "https://docs.litellm.ai/docs/providers/sail", + "endpoints": { + "chat_completions": true, + "messages": false, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false + } + }, "sambanova": { "display_name": "Sambanova (`sambanova`)", "url": "https://docs.litellm.ai/docs/providers/sambanova", diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index 0ca08cb7992..d6f07a5282b 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -2949,6 +2949,34 @@ ], "default_model_placeholder": "gpt-3.5-turbo" }, + { + "provider": "Sail", + "provider_display_name": "Sail", + "litellm_provider": "sail", + "credential_fields": [ + { + "key": "api_base", + "label": "API Base", + "placeholder": "https://api.sailresearch.com/v1", + "tooltip": null, + "required": false, + "field_type": "text", + "options": null, + "default_value": null + }, + { + "key": "api_key", + "label": "API Key", + "placeholder": null, + "tooltip": null, + "required": true, + "field_type": "password", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "sail/zai-org/GLM-5.3" + }, { "provider": "Sambanova", "provider_display_name": "Sambanova", diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 3f8471cbde1..2ec5e703a2f 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -4297,6 +4297,7 @@ class LlmProviders(str, Enum): PINSTRIPES = "pinstripes" COGNITION = "cognition" SCX_AI = "scx-ai" + SAIL = "sail" DARKBLOOM = "darkbloom" META = "meta" LITELLM_AGENT = "litellm_agent" diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ebea1a6044d..1fc08c1175f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -57945,31 +57945,202 @@ "supports_reasoning": false, "source": "https://pinstripes.io/" }, - "pinstripes/ps/deepseek-v4-flash": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 1e-07, - "output_cost_per_token": 2e-07, - "litellm_provider": "pinstripes", + "sail/moonshotai/Kimi-K3": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 2.5e-06, + "output_cost_per_token": 1.25e-05, + "cache_read_input_token_cost": 2.5e-07, + "litellm_provider": "sail", "mode": "chat", "supports_function_calling": true, - "supports_assistant_prefill": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, "supports_reasoning": true, - "source": "https://pinstripes.io/" + "source": "https://docs.sailresearch.com/models" }, - "pinstripes/ps/minimax-m2.7": { - "max_tokens": 1000192, - "max_input_tokens": 1000192, - "max_output_tokens": 1000192, - "input_cost_per_token": 2.55e-07, - "output_cost_per_token": 5.5e-07, - "litellm_provider": "pinstripes", + "sail/zai-org/GLM-5.3": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 9.8e-07, + "output_cost_per_token": 3.08e-06, + "cache_read_input_token_cost": 1.8e-07, + "litellm_provider": "sail", "mode": "chat", "supports_function_calling": true, - "supports_assistant_prefill": true, - "supports_reasoning": false, - "source": "https://pinstripes.io/" + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/zai-org/GLM-5.3-Flash": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 1.1e-07, + "output_cost_per_token": 3.5e-07, + "cache_read_input_token_cost": 2e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/deepseek-ai/DeepSeek-V4-Pro-0813": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 9.2e-07, + "output_cost_per_token": 2.77e-06, + "cache_read_input_token_cost": 4e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/deepseek-ai/DeepSeek-V4-Flash-0731": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 1.8e-07, + "cache_read_input_token_cost": 2e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/deepseek-ai/DeepSeek-V4.1-Flash": { + "max_tokens": 1048576, + "max_input_tokens": 1048576, + "max_output_tokens": 1048576, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 6e-09, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/moonshotai/Kimi-K2.6": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 4e-06, + "cache_read_input_token_cost": 2e-07, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/google/gemma-4-31B-it": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 4e-07, + "output_cost_per_token": 6e-07, + "cache_read_input_token_cost": 2e-07, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/nvidia/Gemma-4-31B-IT-NVFP4": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1.4e-07, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 7e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/google/gemma-4-12B-it": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 1.5e-07, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/openai/gpt-oss-120b": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 3e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "source": "https://docs.sailresearch.com/models" + }, + "sail/Qwen/Qwen3.6-35B-A3B": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 2e-08, + "litellm_provider": "sail", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_vision": true, + "source": "https://docs.sailresearch.com/models" }, "darkbloom/gemma-4-26b": { "input_cost_per_token": 3e-08, diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index b8d1621cde3..49902c17736 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -2263,6 +2263,23 @@ "search": true } }, + "sail": { + "display_name": "Sail (`sail`)", + "url": "https://docs.litellm.ai/docs/providers/sail", + "endpoints": { + "chat_completions": true, + "messages": false, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": false + } + }, "sambanova": { "display_name": "Sambanova (`sambanova`)", "url": "https://docs.litellm.ai/docs/providers/sambanova", diff --git a/tests/test_litellm/llms/openai_like/test_sail_provider.py b/tests/test_litellm/llms/openai_like/test_sail_provider.py new file mode 100644 index 00000000000..430e720f25b --- /dev/null +++ b/tests/test_litellm/llms/openai_like/test_sail_provider.py @@ -0,0 +1,261 @@ +""" +Tests for the Sail (sailresearch.com) JSON-configured provider. + +Each test asserts the outbound HTTP request that litellm would send to Sail, +via a mocked httpx transport, rather than asserting registry contents. +""" + +import json + +import httpx +import pytest +import respx + +import litellm + +SAIL_BASE_URL = "https://api.sailresearch.com/v1" +SAIL_CHAT_COMPLETIONS = f"{SAIL_BASE_URL}/chat/completions" +SAIL_RESPONSES = f"{SAIL_BASE_URL}/responses" + +MODEL = "sail/zai-org/GLM-5.3" + + +def _chat_completion_payload() -> dict: + return { + "id": "chatcmpl-sail", + "object": "chat.completion", + "created": 1234567890, + "model": "zai-org/GLM-5.3", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "sail response"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 2, "completion_tokens": 2, "total_tokens": 4}, + } + + +def _chat_completion_stream() -> str: + chunks = [ + { + "id": "chatcmpl-sail-stream", + "object": "chat.completion.chunk", + "created": 1234567890, + "model": "zai-org/GLM-5.3", + "choices": [ + { + "index": 0, + "delta": {"role": "assistant", "content": "sail"}, + "finish_reason": None, + } + ], + }, + { + "id": "chatcmpl-sail-stream", + "object": "chat.completion.chunk", + "created": 1234567890, + "model": "zai-org/GLM-5.3", + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + }, + ] + return "".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks) + "data: [DONE]\n\n" + + +def _responses_payload() -> dict: + return { + "id": "resp_sail", + "object": "response", + "created_at": 1234567890, + "status": "completed", + "model": "zai-org/GLM-5.3", + "output": [], + "parallel_tool_calls": True, + "usage": {"input_tokens": 2, "output_tokens": 2, "total_tokens": 4}, + "error": None, + } + + +@pytest.fixture(autouse=True) +def _sail_env(monkeypatch: pytest.MonkeyPatch): + litellm.disable_aiohttp_transport = True + monkeypatch.setenv("SAIL_API_KEY", "sk-sail-test") + monkeypatch.delenv("SAIL_API_BASE", raising=False) + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + +class TestSailRequestShape: + @pytest.mark.respx() + def test_sail_chat_completions_url_and_auth(self, respx_mock: respx.Router): + respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload()) + + litellm.completion( + model=MODEL, + messages=[{"role": "user", "content": "hi"}], + ) + + assert len(respx_mock.calls) == 1 + request = respx_mock.calls[0].request + assert request.url == SAIL_CHAT_COMPLETIONS + assert request.headers["Authorization"] == "Bearer sk-sail-test" + + from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider + + _, provider, _, _ = get_llm_provider( + model=MODEL, custom_llm_provider=None, api_base=None, api_key=None + ) + assert provider == "sail" + + @pytest.mark.respx() + def test_sail_api_base_env_overrides_url( + self, respx_mock: respx.Router, monkeypatch: pytest.MonkeyPatch + ): + monkeypatch.setenv("SAIL_API_BASE", "https://sail.internal.example/v2") + route = respx_mock.post("https://sail.internal.example/v2/chat/completions").respond( + json=_chat_completion_payload() + ) + + litellm.completion(model=MODEL, messages=[{"role": "user", "content": "hi"}]) + + assert route.called + + @pytest.mark.respx() + def test_sail_explicit_api_base_trailing_slash_no_double_slash(self, respx_mock: respx.Router): + route = respx_mock.post("https://custom.sail.example/v1/chat/completions").respond( + json=_chat_completion_payload() + ) + + litellm.completion( + model=MODEL, + messages=[{"role": "user", "content": "hi"}], + api_base="https://custom.sail.example/v1/", + ) + + assert route.called + + @pytest.mark.respx() + def test_sail_max_tokens_sent_as_max_completion_tokens(self, respx_mock: respx.Router): + respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload()) + + litellm.completion( + model=MODEL, + messages=[{"role": "user", "content": "hi"}], + max_tokens=100, + ) + + body = json.loads(respx_mock.calls[0].request.content) + assert body["max_completion_tokens"] == 100 + assert "max_tokens" not in body + + @pytest.mark.respx() + def test_sail_no_metadata_key_when_caller_passes_none(self, respx_mock: respx.Router): + respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload()) + + litellm.completion(model=MODEL, messages=[{"role": "user", "content": "hi"}]) + + body = json.loads(respx_mock.calls[0].request.content) + assert "metadata" not in body + + @pytest.mark.respx() + @pytest.mark.parametrize("stream", [False, True], ids=["non-streaming", "streaming"]) + def test_sail_completion_window_via_extra_body(self, respx_mock: respx.Router, stream: bool): + if stream: + respx_mock.post(SAIL_CHAT_COMPLETIONS).respond( + content=_chat_completion_stream(), + headers={"content-type": "text/event-stream"}, + ) + else: + respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload()) + + response = litellm.completion( + model=MODEL, + messages=[{"role": "user", "content": "hi"}], + stream=stream, + extra_body={"metadata": {"completion_window": "balanced"}}, + ) + if stream: + list(response) + + body = json.loads(respx_mock.calls[0].request.content) + assert body["metadata"] == {"completion_window": "balanced"} + + @pytest.mark.respx() + def test_sail_tools_survive_to_request_body(self, respx_mock: respx.Router): + respx_mock.post(SAIL_CHAT_COMPLETIONS).respond(json=_chat_completion_payload()) + + litellm.completion( + model=MODEL, + messages=[{"role": "user", "content": "what is the weather"}], + tools=[ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the weather for a city", + "parameters": { + "type": "object", + "properties": {"city": {"type": "string"}}, + "required": ["city"], + }, + }, + } + ], + ) + + body = json.loads(respx_mock.calls[0].request.content) + assert body["tools"][0]["function"]["name"] == "get_weather" + + @pytest.mark.asyncio + @pytest.mark.respx() + async def test_sail_aresponses_posts_metadata_and_background(self, respx_mock: respx.Router): + respx_mock.post(SAIL_RESPONSES).respond(json=_responses_payload()) + + await litellm.aresponses( + model=MODEL, + input="hi", + metadata={"completion_window": "flex"}, + background=True, + ) + + assert len(respx_mock.calls) == 1 + request = respx_mock.calls[0].request + assert request.url == SAIL_RESPONSES + body = json.loads(request.content) + assert body["metadata"] == {"completion_window": "flex"} + assert body["background"] is True + + +class TestSailCostTracking: + def test_cached_tokens_billed_at_sail_cache_read_rate(self, monkeypatch: pytest.MonkeyPatch): + from litellm.types.utils import PromptTokensDetailsWrapper, Usage + + rates = litellm.model_cost[MODEL] + prompt_tokens = 1000 + cached_tokens = 600 + completion_tokens = 200 + + response = litellm.ModelResponse( + model="zai-org/GLM-5.3", + choices=[{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}], + usage=Usage( + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + total_tokens=prompt_tokens + completion_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens), + ), + ) + + cost = litellm.completion_cost( + completion_response=response, + model=MODEL, + custom_llm_provider="sail", + ) + + expected = ( + (prompt_tokens - cached_tokens) * rates["input_cost_per_token"] + + cached_tokens * rates["cache_read_input_token_cost"] + + completion_tokens * rates["output_cost_per_token"] + ) + assert cost == pytest.approx(expected)