diff --git a/litellm/__init__.py b/litellm/__init__.py index 64c60ca3374..454f9790923 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1639,6 +1639,9 @@ if TYPE_CHECKING: from .llms.perplexity.responses.transformation import ( PerplexityResponsesConfig as PerplexityResponsesConfig, ) + from .llms.sail_research.responses.transformation import ( + SailResearchResponsesConfig as SailResearchResponsesConfig, + ) from .llms.databricks.responses.transformation import ( DatabricksResponsesAPIConfig as DatabricksResponsesAPIConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index 9164a3c8ae4..db87261bf11 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -232,6 +232,7 @@ LLM_CONFIG_NAMES = ( "HostedVLLMResponsesAPIConfig", "VolcEngineResponsesAPIConfig", "PerplexityResponsesConfig", + "SailResearchResponsesConfig", "DatabricksResponsesAPIConfig", "OpenRouterResponsesAPIConfig", "GoogleAIStudioInteractionsConfig", @@ -933,6 +934,10 @@ _LLM_CONFIGS_IMPORT_MAP = { ".llms.perplexity.responses.transformation", "PerplexityResponsesConfig", ), + "SailResearchResponsesConfig": ( + ".llms.sail_research.responses.transformation", + "SailResearchResponsesConfig", + ), "DatabricksResponsesAPIConfig": ( ".llms.databricks.responses.transformation", "DatabricksResponsesAPIConfig", diff --git a/litellm/constants.py b/litellm/constants.py index 337cb1243fb..60e48902235 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -807,6 +807,7 @@ openai_compatible_providers: List = [ "clarifai", "docker_model_runner", "ragflow", + "sail_research", ] openai_text_completion_compatible_providers: List = ( [ # providers that support `/v1/completions` diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 95bcd4d7186..7b773589079 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -532,7 +532,10 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 ) return model, custom_llm_provider, dynamic_api_key, api_base - if custom_llm_provider == "perplexity": + if custom_llm_provider == "sail_research": + api_base = api_base or get_secret_str("SAIL_API_BASE") or "https://api.sailresearch.com" + dynamic_api_key = api_key or get_secret_str("SAIL_API_KEY") + elif custom_llm_provider == "perplexity": # perplexity is openai compatible, we just need to set this to custom_openai and have the api_base be https://api.perplexity.ai ( api_base, diff --git a/litellm/llms/sail_research/__init__.py b/litellm/llms/sail_research/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/litellm/llms/sail_research/responses/__init__.py b/litellm/llms/sail_research/responses/__init__.py new file mode 100644 index 00000000000..a46d6085014 --- /dev/null +++ b/litellm/llms/sail_research/responses/__init__.py @@ -0,0 +1,7 @@ +""" +Sail Research Responses API module +""" + +from .transformation import SailResearchResponsesConfig + +__all__ = ["SailResearchResponsesConfig"] diff --git a/litellm/llms/sail_research/responses/transformation.py b/litellm/llms/sail_research/responses/transformation.py new file mode 100644 index 00000000000..449086e3109 --- /dev/null +++ b/litellm/llms/sail_research/responses/transformation.py @@ -0,0 +1,109 @@ +""" +Sail Responses API — OpenAI-compatible subset. + +Sail is a responses-only provider. Key differences from vanilla OpenAI: +- No streaming (stream: true is rejected) +- No instructions, previous_response_id, conversation, prompt params +- No server-side tools (web_search, file_search, code_interpreter, etc.) +- No parallel_tool_calls +- No text.format.type "json_object" (use json_schema instead) +- background: true is supported + +Ref: https://api.sailresearch.com +""" + +from typing import Any, Dict, List, Optional, Union + +import httpx + +from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj +from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig +from litellm.secret_managers.main import get_secret_str +from litellm.types.llms.openai import ResponseInputParam +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import LlmProviders + + +class SailResearchResponsesConfig(OpenAIResponsesAPIConfig): + def get_supported_openai_params(self, model: str) -> list: + return [ + "model", + "input", + "temperature", + "top_p", + "max_output_tokens", + "tools", + "tool_choice", + "text", + "reasoning", + "metadata", + "background", + "extra_headers", + "extra_query", + "extra_body", + "timeout", + ] + + @property + def custom_llm_provider(self) -> LlmProviders: + return LlmProviders.SAIL_RESEARCH + + def should_fake_stream( + self, + model: Optional[str] = None, + stream: Optional[bool] = None, + custom_llm_provider: Optional[str] = None, + ) -> bool: + """Sail does not support streaming — fake it client-side.""" + return True + + def validate_environment( + self, headers: dict, model: str, litellm_params: Optional[GenericLiteLLMParams] + ) -> dict: + litellm_params = litellm_params or GenericLiteLLMParams() + api_key = litellm_params.api_key or get_secret_str("SAIL_API_KEY") + if api_key: + headers["Authorization"] = f"Bearer {api_key}" + return headers + + def get_complete_url(self, api_base: Optional[str], litellm_params: dict) -> str: + api_base = ( + api_base + or get_secret_str("SAIL_API_BASE") + or "https://api.sailresearch.com" + ) + return f"{api_base.rstrip('/')}/v1/responses" + + def transform_responses_api_request( + self, + model: str, + input: Union[str, ResponseInputParam], + response_api_optional_request_params: Dict, + litellm_params: GenericLiteLLMParams, + headers: dict, + ) -> Dict: + # Strip json_object format type — Sail only accepts json_schema + text_config = response_api_optional_request_params.get("text") + if isinstance(text_config, dict): + fmt = text_config.get("format") + if isinstance(fmt, dict) and fmt.get("type") == "json_object": + text_config = {**text_config} + del text_config["format"] + response_api_optional_request_params = { + **response_api_optional_request_params, + "text": text_config, + } + + # Never send stream: true to Sail + response_api_optional_request_params.pop("stream", None) + + return super().transform_responses_api_request( + model=model, + input=input, + response_api_optional_request_params=response_api_optional_request_params, + litellm_params=litellm_params, + headers=headers, + ) + + def supports_native_websocket(self) -> bool: + return False diff --git a/litellm/types/utils.py b/litellm/types/utils.py index cd5806b3ab7..b85a1b04d06 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3296,6 +3296,7 @@ class LlmProviders(str, Enum): LITELLM_AGENT = "litellm_agent" CURSOR = "cursor" BEDROCK_MANTLE = "bedrock_mantle" + SAIL_RESEARCH = "sail_research" # Create a set of all provider values for quick lookup diff --git a/litellm/utils.py b/litellm/utils.py index f902644e760..2372e849920 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8538,6 +8538,8 @@ class ProviderConfigManager: return litellm.ManusResponsesAPIConfig() elif litellm.LlmProviders.PERPLEXITY == provider: return litellm.PerplexityResponsesConfig() + elif litellm.LlmProviders.SAIL_RESEARCH == provider: + return litellm.SailResearchResponsesConfig() elif litellm.LlmProviders.DATABRICKS == provider: # Databricks Responses API is only compatible with OpenAI GPT models if model and "gpt" in model.lower(): diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 90ff7d1103c..9e8ee6ef623 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -38188,5 +38188,55 @@ "tool_use_system_prompt_tokens": 346, "supports_native_structured_output": true, "supports_pdf_input": true + }, + "sail_research/moonshotai/Kimi-K2.5": { + "litellm_provider": "sail_research", + "mode": "responses", + "supported_endpoints": [ + "/v1/responses" + ], + "supports_function_calling": true + }, + "sail_research/zai-org/GLM-5": { + "litellm_provider": "sail_research", + "mode": "responses", + "max_input_tokens": 202752, + "supported_endpoints": [ + "/v1/responses" + ], + "supports_function_calling": true + }, + "sail_research/zai-org/GLM-5.1-FP8": { + "litellm_provider": "sail_research", + "mode": "responses", + "max_input_tokens": 202752, + "supported_endpoints": [ + "/v1/responses" + ], + "supports_function_calling": true + }, + "sail_research/deepseek-ai/DeepSeek-V3.2": { + "litellm_provider": "sail_research", + "mode": "responses", + "supported_endpoints": [ + "/v1/responses" + ], + "supports_function_calling": true + }, + "sail_research/openai/gpt-oss-20b": { + "litellm_provider": "sail_research", + "mode": "responses", + "supported_endpoints": [ + "/v1/responses" + ], + "supports_function_calling": true + }, + "sail_research/openai/gpt-oss-120b": { + "litellm_provider": "sail_research", + "mode": "responses", + "supported_endpoints": [ + "/v1/responses" + ], + "supports_function_calling": true } } \ No newline at end of file diff --git a/tests/test_litellm/llms/sail_research/__init__.py b/tests/test_litellm/llms/sail_research/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/sail_research/responses/__init__.py b/tests/test_litellm/llms/sail_research/responses/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/sail_research/responses/test_sail_research_responses_transformation.py b/tests/test_litellm/llms/sail_research/responses/test_sail_research_responses_transformation.py new file mode 100644 index 00000000000..d10f3bfd63f --- /dev/null +++ b/tests/test_litellm/llms/sail_research/responses/test_sail_research_responses_transformation.py @@ -0,0 +1,293 @@ +""" +Tests for Sail Research Responses API transformation + +Tests the SailResearchResponsesConfig class that handles Sail-specific +transformations for the Responses API. + +Source: litellm/llms/sail_research/responses/transformation.py +""" + +import os +import sys + +import httpx +import pytest + +sys.path.insert(0, os.path.abspath("../../../../..")) + +from litellm.llms.sail_research.responses.transformation import ( + SailResearchResponsesConfig, +) +from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + + +class TestSailResearchResponsesTransformation: + """Test Sail Research Responses API configuration and transformations""" + + def test_supported_params(self): + """get_supported_openai_params returns Sail-specific restricted list""" + config = SailResearchResponsesConfig() + supported = config.get_supported_openai_params( + "sail_research/deepseek-ai/DeepSeek-V3.2" + ) + + expected = [ + "model", + "input", + "temperature", + "top_p", + "max_output_tokens", + "tools", + "tool_choice", + "text", + "reasoning", + "metadata", + "background", + ] + + for param in expected: + assert param in supported, f"Missing supported param: {param}" + + # Params that Sail does NOT support + unsupported = [ + "instructions", + "previous_response_id", + "parallel_tool_calls", + "store", + "truncation", + ] + + for param in unsupported: + assert param not in supported, f"Param should not be supported: {param}" + + def test_should_fake_stream(self): + """Sail does not support streaming — should_fake_stream must return True""" + config = SailResearchResponsesConfig() + assert config.should_fake_stream() is True + assert ( + config.should_fake_stream(model="deepseek-ai/DeepSeek-V3.2", stream=True) + is True + ) + + def test_custom_llm_provider(self): + """Provider enum is SAIL_RESEARCH""" + config = SailResearchResponsesConfig() + assert config.custom_llm_provider == LlmProviders.SAIL_RESEARCH + + def test_get_complete_url_default(self): + """Default URL points to Sail Research API""" + config = SailResearchResponsesConfig() + url = config.get_complete_url(api_base=None, litellm_params={}) + assert url == "https://api.sailresearch.com/v1/responses" + + def test_get_complete_url_custom(self): + """Custom api_base is respected""" + config = SailResearchResponsesConfig() + url = config.get_complete_url( + api_base="https://custom.sail.dev", litellm_params={} + ) + assert url == "https://custom.sail.dev/v1/responses" + + def test_get_complete_url_trailing_slash(self): + """Trailing slash in api_base is handled""" + config = SailResearchResponsesConfig() + url = config.get_complete_url( + api_base="https://api.sailresearch.com/", litellm_params={} + ) + assert url == "https://api.sailresearch.com/v1/responses" + + def test_no_double_v1_in_default_url(self): + """Regression: default api_base must not produce /v1/v1/responses""" + config = SailResearchResponsesConfig() + # Simulate the default base URL that get_llm_provider_logic resolves + url = config.get_complete_url( + api_base="https://api.sailresearch.com", litellm_params={} + ) + assert url == "https://api.sailresearch.com/v1/responses" + assert "/v1/v1/" not in url + + def test_stream_stripped_from_request(self): + """stream param is removed from the request body""" + config = SailResearchResponsesConfig() + + data = config.transform_responses_api_request( + model="deepseek-ai/DeepSeek-V3.2", + input="Hello", + response_api_optional_request_params={"stream": True, "temperature": 0.7}, + litellm_params={}, + headers={}, + ) + + assert "stream" not in data + assert data["temperature"] == 0.7 + + def test_json_object_format_stripped(self): + """text.format.type json_object is stripped (Sail only accepts json_schema)""" + config = SailResearchResponsesConfig() + + data = config.transform_responses_api_request( + model="deepseek-ai/DeepSeek-V3.2", + input="Return JSON", + response_api_optional_request_params={ + "text": { + "format": {"type": "json_object"}, + }, + }, + litellm_params={}, + headers={}, + ) + + # The json_object format should be stripped + text_config = data.get("text", {}) + assert "format" not in text_config + + def test_json_schema_format_preserved(self): + """text.format.type json_schema is preserved""" + config = SailResearchResponsesConfig() + + schema_format = { + "type": "json_schema", + "name": "my_schema", + "schema": {"type": "object", "properties": {"x": {"type": "number"}}}, + } + + data = config.transform_responses_api_request( + model="deepseek-ai/DeepSeek-V3.2", + input="Return JSON", + response_api_optional_request_params={ + "text": {"format": schema_format}, + }, + litellm_params={}, + headers={}, + ) + + assert data["text"]["format"]["type"] == "json_schema" + + def test_function_tool_passthrough(self): + """Function tools are preserved""" + config = SailResearchResponsesConfig() + + params = ResponsesAPIOptionalRequestParams( + tools=[ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get weather", + "parameters": {"type": "object", "properties": {}}, + }, + } + ] + ) + + result = config.map_openai_params( + response_api_optional_params=params, + model="sail_research/deepseek-ai/DeepSeek-V3.2", + drop_params=False, + ) + + assert "tools" in result + assert result["tools"][0]["type"] == "function" + assert result["tools"][0]["function"]["name"] == "get_weather" + + def test_background_param_supported(self): + """background: true is a supported Sail-specific param""" + config = SailResearchResponsesConfig() + + params = ResponsesAPIOptionalRequestParams(background=True) + + result = config.map_openai_params( + response_api_optional_params=params, + model="sail_research/deepseek-ai/DeepSeek-V3.2", + drop_params=False, + ) + + assert result.get("background") is True + + def test_provider_config_registration(self): + """ProviderConfigManager returns SailResearchResponsesConfig""" + config = ProviderConfigManager.get_provider_responses_api_config( + model="sail_research/deepseek-ai/DeepSeek-V3.2", + provider=LlmProviders.SAIL_RESEARCH, + ) + + assert config is not None + assert isinstance(config, SailResearchResponsesConfig) + assert config.custom_llm_provider == LlmProviders.SAIL_RESEARCH + + def test_no_native_websocket(self): + """Sail does not support native WebSocket""" + config = SailResearchResponsesConfig() + assert config.supports_native_websocket() is False + + def test_validate_environment_sets_auth_header(self): + """API key is set in Authorization header""" + config = SailResearchResponsesConfig() + from litellm.types.router import GenericLiteLLMParams + + headers = config.validate_environment( + headers={}, + model="deepseek-ai/DeepSeek-V3.2", + litellm_params=GenericLiteLLMParams(api_key="sk-test-123"), + ) + + assert headers["Authorization"] == "Bearer sk-test-123" + + def test_successful_response_passes_through(self): + """Normal completed response delegates to base OpenAI handler""" + from litellm.litellm_core_utils.litellm_logging import ( + Logging as LiteLLMLoggingObj, + ) + + config = SailResearchResponsesConfig() + + success_body = { + "id": "resp_123", + "object": "response", + "created_at": 1700000000, + "status": "completed", + "model": "deepseek-ai/DeepSeek-V3.2", + "output": [ + { + "type": "message", + "id": "msg_123", + "role": "assistant", + "status": "completed", + "content": [ + {"type": "output_text", "text": "Hello!", "annotations": []} + ], + } + ], + "usage": { + "input_tokens": 10, + "output_tokens": 5, + "total_tokens": 15, + }, + } + + raw_response = httpx.Response( + status_code=200, + json=success_body, + request=httpx.Request("POST", "https://api.sailresearch.com/v1/responses"), + ) + + logging_obj = LiteLLMLoggingObj( + model="sail_research/deepseek-ai/DeepSeek-V3.2", + messages=[], + stream=False, + call_type="responses", + start_time=None, + litellm_call_id="test", + function_id="test", + ) + + response = config.transform_response_api_response( + model="sail_research/deepseek-ai/DeepSeek-V3.2", + raw_response=raw_response, + logging_obj=logging_obj, + ) + + assert response.id == "resp_123" + assert response.status == "completed"