diff --git a/docs/my-website/docs/providers/minimax_tts.md b/docs/my-website/docs/providers/minimax_tts.md new file mode 100644 index 00000000000..314d705413b --- /dev/null +++ b/docs/my-website/docs/providers/minimax_tts.md @@ -0,0 +1,281 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# MiniMax - Text-to-Speech + +## Overview + +MiniMax provides high-quality text-to-speech synthesis with support for 40+ languages and ultra-low latency. LiteLLM provides a unified OpenAI-compatible interface for MiniMax TTS. + +| Feature | Supported | Notes | +|---------|-----------|-------| +| Logging | ✅ | Works across all integrations | +| Fallbacks | ✅ | Works between supported models | +| Loadbalancing | ✅ | Works between supported models | +| Guardrails | ✅ | Applies to input text | +| Supported Models | speech-2.6-hd, speech-2.6-turbo, speech-02-hd, speech-02-turbo | | + +## Supported Models + +| Model | Description | +|-------|-------------| +| speech-2.6-hd | Ultra-low latency, intelligence parsing, and enhanced naturalness | +| speech-2.6-turbo | Faster, more affordable, ideal for agents | +| speech-02-hd | Superior rhythm and stability with outstanding replication similarity | +| speech-02-turbo | Superior rhythm and stability with enhanced multilingual capabilities | +| speech-01-hd | Previous generation HD model | +| speech-01-turbo | Previous generation turbo model | + +## Quick Start + +## **LiteLLM Python SDK Usage** + +### Basic Usage + +```python +from pathlib import Path +from litellm import speech +import os + +os.environ["MINIMAX_API_KEY"] = "your-api-key" + +speech_file_path = Path(__file__).parent / "speech.mp3" +response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="The quick brown fox jumped over the lazy dogs", +) +response.stream_to_file(speech_file_path) +``` + +### Async Usage + +```python +from litellm import aspeech +from pathlib import Path +import os, asyncio + +os.environ["MINIMAX_API_KEY"] = "your-api-key" + +async def test_async_speech(): + speech_file_path = Path(__file__).parent / "speech.mp3" + response = await aspeech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="The quick brown fox jumped over the lazy dogs", + ) + response.stream_to_file(speech_file_path) + +asyncio.run(test_async_speech()) +``` + +### Voice Selection + +MiniMax supports many voices. LiteLLM provides OpenAI-compatible voice names that map to MiniMax voices: + +```python +from litellm import speech + +# OpenAI-compatible voice names +voices = ["alloy", "echo", "fable", "onyx", "nova", "shimmer"] + +for voice in voices: + response = speech( + model="minimax/speech-2.6-hd", + voice=voice, + input=f"This is the {voice} voice", + ) + response.stream_to_file(f"speech_{voice}.mp3") +``` + +You can also use MiniMax-native voice IDs directly: + +```python +response = speech( + model="minimax/speech-2.6-hd", + voice="male-qn-qingse", # MiniMax native voice ID + input="Using native MiniMax voice ID", +) +``` + +### Custom Parameters + +MiniMax TTS supports additional parameters for fine-tuning audio output: + +```python +from litellm import speech + +response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="Custom audio parameters", + speed=1.5, # Speed: 0.5 to 2.0 + response_format="mp3", # Format: mp3, pcm, wav, flac + extra_body={ + "vol": 1.2, # Volume: 0.1 to 10 + "pitch": 2, # Pitch adjustment: -12 to 12 + "sample_rate": 32000, # 16000, 24000, or 32000 + "bitrate": 128000, # For MP3: 64000, 128000, 192000, 256000 + "channel": 1, # 1 for mono, 2 for stereo + } +) +response.stream_to_file("custom_speech.mp3") +``` + +### Response Formats + +```python +from litellm import speech + +# MP3 format (default) +response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="MP3 format audio", + response_format="mp3", +) + +# PCM format +response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="PCM format audio", + response_format="pcm", +) + +# WAV format +response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="WAV format audio", + response_format="wav", +) + +# FLAC format +response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="FLAC format audio", + response_format="flac", +) +``` + +## **LiteLLM Proxy Usage** + +LiteLLM provides an OpenAI-compatible `/audio/speech` endpoint for MiniMax TTS. + +### Setup + +Add MiniMax to your proxy configuration: + +```yaml +model_list: + - model_name: tts + litellm_params: + model: minimax/speech-2.6-hd + api_key: os.environ/MINIMAX_API_KEY + + - model_name: tts-turbo + litellm_params: + model: minimax/speech-2.6-turbo + api_key: os.environ/MINIMAX_API_KEY +``` + +Start the proxy: + +```bash +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +### Making Requests + +```bash +curl http://0.0.0.0:4000/v1/audio/speech \ + -H "Authorization: Bearer sk-1234" \ + -H "Content-Type: application/json" \ + -d '{ + "model": "tts", + "input": "The quick brown fox jumped over the lazy dog.", + "voice": "alloy" + }' \ + --output speech.mp3 +``` + +With custom parameters: + +```bash +curl http://0.0.0.0:4000/v1/audio/speech \ + -H "Authorization: Bearer sk-1234" \ + -H "Content-Type: application/json" \ + -d '{ + "model": "tts", + "input": "Custom parameters example.", + "voice": "nova", + "speed": 1.5, + "response_format": "mp3", + "extra_body": { + "vol": 1.2, + "pitch": 1, + "sample_rate": 32000 + } + }' \ + --output custom_speech.mp3 +``` + +## Voice Mappings + +LiteLLM maps OpenAI-compatible voice names to MiniMax voice IDs: + +| OpenAI Voice | MiniMax Voice ID | Description | +|--------------|------------------|-------------| +| alloy | male-qn-qingse | Male voice | +| echo | male-qn-jingying | Male voice | +| fable | female-shaonv | Female voice | +| onyx | male-qn-badao | Male voice | +| nova | female-yujie | Female voice | +| shimmer | female-tianmei | Female voice | + +You can also use any MiniMax-native voice ID directly by passing it as the `voice` parameter. + + +### Streaming (WebSocket) + +:::note +The current implementation uses MiniMax's HTTP endpoint. For WebSocket streaming support, please refer to MiniMax's official documentation at [https://platform.minimax.io/docs](https://platform.minimax.io/docs). +::: + +## Error Handling + +```python +from litellm import speech +import litellm + +try: + response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="Test input", + ) + response.stream_to_file("output.mp3") +except litellm.exceptions.BadRequestError as e: + print(f"Bad request: {e}") +except litellm.exceptions.AuthenticationError as e: + print(f"Authentication failed: {e}") +except Exception as e: + print(f"Error: {e}") +``` + +### Extra Body Parameters + +Pass these via `extra_body`: + +| Parameter | Type | Description | Default | +|-----------|------|-------------|---------| +| vol | float | Volume (0.1 to 10) | 1.0 | +| pitch | int | Pitch adjustment (-12 to 12) | 0 | +| sample_rate | int | Sample rate: 16000, 24000, 32000 | 32000 | +| bitrate | int | Bitrate for MP3: 64000, 128000, 192000, 256000 | 128000 | +| channel | int | Audio channels: 1 (mono) or 2 (stereo) | 1 | +| output_format | string | Output format: "hex" or "url" (url returns a URL valid for 24 hours) | hex | diff --git a/docs/my-website/docs/text_to_speech.md b/docs/my-website/docs/text_to_speech.md index ce298b538df..f5788630949 100644 --- a/docs/my-website/docs/text_to_speech.md +++ b/docs/my-website/docs/text_to_speech.md @@ -14,7 +14,7 @@ import TabItem from '@theme/TabItem'; | Fallbacks | ✅ | Works between supported models | | Loadbalancing | ✅ | Works between supported models | | Guardrails | ✅ | Applies to input text (non-streaming only) | -| Supported Providers | OpenAI, Azure OpenAI, Vertex AI, AWS Polly, ElevenLabs | | +| Supported Providers | OpenAI, Azure OpenAI, Vertex AI, AWS Polly, ElevenLabs , MiniMax| | ## **LiteLLM Python SDK Usage** ### Quick Start @@ -105,6 +105,7 @@ litellm --config /path/to/config.yaml | Vertex AI | [Usage](../docs/providers/vertex#text-to-speech-apis) | | Gemini | [Usage](#gemini-text-to-speech) | | ElevenLabs | [Usage](../docs/providers/elevenlabs#text-to-speech-tts) | +| MiniMax | [Usage](../docs/providers/minimax_tts) | ## `/audio/speech` to `/chat/completions` Bridge diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 9801c764acc..ae63bade025 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -723,6 +723,7 @@ const sidebars = { "providers/milvus_vector_stores", "providers/mistral", "providers/minimax", + "providers/minimax_tts", "providers/moonshot", "providers/morph", "providers/nebius", diff --git a/litellm/llms/minimax/__init__.py b/litellm/llms/minimax/__init__.py new file mode 100644 index 00000000000..19093c2dadb --- /dev/null +++ b/litellm/llms/minimax/__init__.py @@ -0,0 +1,14 @@ +""" +MiniMax LLM Provider +""" + +from .text_to_speech.transformation import ( + MinimaxException, + MinimaxTextToSpeechConfig, +) + +__all__ = [ + "MinimaxTextToSpeechConfig", + "MinimaxException", +] + diff --git a/litellm/llms/minimax/text_to_speech/__init__.py b/litellm/llms/minimax/text_to_speech/__init__.py new file mode 100644 index 00000000000..e3fcddeb05f --- /dev/null +++ b/litellm/llms/minimax/text_to_speech/__init__.py @@ -0,0 +1,8 @@ +""" +MiniMax Text-to-Speech module +""" + +from .transformation import MinimaxException, MinimaxTextToSpeechConfig + +__all__ = ["MinimaxTextToSpeechConfig", "MinimaxException"] + diff --git a/litellm/llms/minimax/text_to_speech/transformation.py b/litellm/llms/minimax/text_to_speech/transformation.py new file mode 100644 index 00000000000..a3a75d220ff --- /dev/null +++ b/litellm/llms/minimax/text_to_speech/transformation.py @@ -0,0 +1,421 @@ +""" +MiniMax Text-to-Speech transformation + +Maps OpenAI TTS spec to MiniMax TTS API (WebSocket-based HTTP API) +Reference: https://platform.minimax.io/docs +""" + +from typing import TYPE_CHECKING, Any, Dict, Optional, Tuple, Union + +import httpx +from httpx import Headers + +import litellm +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.base_llm.text_to_speech.transformation import ( + BaseTextToSpeechConfig, + TextToSpeechRequestData, +) +from litellm.secret_managers.main import get_secret_str + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + from litellm.types.llms.openai import HttpxBinaryResponseContent +else: + LiteLLMLoggingObj = Any + HttpxBinaryResponseContent = Any + + +class MinimaxException(BaseLLMException): + """Custom exception for MiniMax API errors""" + + def __init__( + self, + status_code: int, + message: str, + headers: Optional[Union[dict, Headers]] = None, + ): + super().__init__(status_code=status_code, message=message, headers=headers) + + +class MinimaxTextToSpeechConfig(BaseTextToSpeechConfig): + """ + Configuration for MiniMax Text-to-Speech + + Reference: https://platform.minimax.io/docs + + MiniMax TTS API supports both WebSocket and HTTP endpoints. + This implementation uses the HTTP endpoint for simplicity. + """ + + TTS_BASE_URL = "https://api.minimax.io" + TTS_ENDPOINT_PATH = "/v1/t2a_v2" + + # Voice mappings from OpenAI-style voices to MiniMax voice IDs + # MiniMax supports many voices, these are common mappings + VOICE_MAPPINGS = { + "alloy": "male-qn-qingse", + "echo": "male-qn-jingying", + "fable": "female-shaonv", + "onyx": "male-qn-badao", + "nova": "female-yujie", + "shimmer": "female-tianmei", + } + + # Response format mappings from OpenAI to MiniMax + FORMAT_MAPPINGS = { + "mp3": "mp3", + "pcm": "pcm", + "wav": "wav", + "flac": "flac", + } + + def get_supported_openai_params(self, model: str) -> list: + """ + MiniMax TTS supports these OpenAI parameters + """ + return ["voice", "response_format", "speed"] + + def _extract_voice_id(self, voice: str) -> str: + """ + Normalize the provided voice information into a MiniMax voice_id. + """ + normalized_voice = voice.strip() + mapped_voice = self.VOICE_MAPPINGS.get(normalized_voice.lower()) + return mapped_voice or normalized_voice + + def _resolve_voice_id( + self, + voice: Optional[Union[str, Dict[str, Any]]], + params: Dict[str, Any], + ) -> str: + """ + Determine the MiniMax voice_id based on provided voice input or parameters. + """ + mapped_voice: Optional[str] = None + + if isinstance(voice, str) and voice.strip(): + mapped_voice = self._extract_voice_id(voice) + elif isinstance(voice, dict): + for key in ("voice_id", "id", "name"): + candidate = voice.get(key) + if isinstance(candidate, str) and candidate.strip(): + mapped_voice = self._extract_voice_id(candidate) + break + elif voice is not None: + mapped_voice = self._extract_voice_id(str(voice)) + + if mapped_voice is None: + voice_override = params.pop("voice_id", None) + if isinstance(voice_override, str) and voice_override.strip(): + mapped_voice = self._extract_voice_id(voice_override) + + if mapped_voice is None: + # Default to a common voice if not specified + mapped_voice = "male-qn-qingse" + + return mapped_voice + + def map_openai_params( + self, + model: str, + optional_params: Dict, + voice: Optional[Union[str, Dict]] = None, + drop_params: bool = False, + kwargs: Optional[Dict[str, Any]] = None, + ) -> Tuple[Optional[str], Dict]: + """ + Map OpenAI parameters to MiniMax TTS parameters + """ + mapped_params: Dict[str, Any] = {} + + # Work on a copy so we don't mutate the caller's dictionary + params = dict(optional_params) if optional_params else {} + + # Extract voice identifier + mapped_voice = self._resolve_voice_id(voice, params) + + # Response/output format + response_format = params.pop("response_format", None) + if isinstance(response_format, str): + mapped_format = self.FORMAT_MAPPINGS.get(response_format, "mp3") + mapped_params["format"] = mapped_format + else: + mapped_params["format"] = "mp3" # Default format + + # Speed parameter (MiniMax supports speed from 0.5 to 2.0) + speed = params.pop("speed", None) + if speed is not None: + try: + speed_value = float(speed) + # Clamp speed to MiniMax's supported range + speed_value = max(0.5, min(2.0, speed_value)) + mapped_params["speed"] = speed_value + except (TypeError, ValueError): + mapped_params["speed"] = 1.0 + else: + mapped_params["speed"] = 1.0 + + # Instructions parameter is OpenAI-specific; omit to prevent API errors + params.pop("instructions", None) + + # Store voice_id for later use in request construction + mapped_params["voice_id"] = mapped_voice + + # Handle extra_body for additional MiniMax-specific parameters + extra_body = params.pop("extra_body", None) + if isinstance(extra_body, dict): + for key, value in extra_body.items(): + if value is not None: + mapped_params[key] = value + + # Pass through any remaining parameters + for key, value in params.items(): + if value is not None: + mapped_params[key] = value + + return mapped_voice, mapped_params + + def validate_environment( + self, + headers: dict, + model: str, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + """ + Validate MiniMax environment and set up authentication headers + """ + api_key = ( + api_key + or litellm.api_key + or get_secret_str("MINIMAX_API_KEY") + ) + + if api_key is None: + raise ValueError( + "MiniMax API key is required. Set MINIMAX_API_KEY environment variable or pass api_key parameter." + ) + + headers.update( + { + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + } + ) + + return headers + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, Headers] + ) -> BaseLLMException: + return MinimaxException( + message=error_message, status_code=status_code, headers=headers + ) + + def transform_text_to_speech_request( + self, + model: str, + input: str, + voice: Optional[str], + optional_params: Dict, + litellm_params: Dict, + headers: dict, + ) -> TextToSpeechRequestData: + """ + Build the MiniMax TTS request payload. + + MiniMax uses a different structure than OpenAI: + - model: The TTS model to use + - text: The input text + - voice_setting: Voice configuration + - audio_setting: Audio output configuration + """ + params = dict(optional_params) if optional_params else {} + + # Extract parameters + voice_id = params.pop("voice_id", voice or "male-qn-qingse") + speed = params.pop("speed", 1.0) + audio_format = params.pop("format", "mp3") + + # Extract additional voice settings + vol = params.pop("vol", 1.0) # Volume (0.1 to 10) + pitch = params.pop("pitch", 0) # Pitch adjustment (-12 to 12) + + # Extract audio settings + sample_rate = params.pop("sample_rate", 32000) # 16000, 24000, 32000 + bitrate = params.pop("bitrate", 128000) # For MP3: 64000, 128000, 192000, 256000 + channel = params.pop("channel", 1) # 1 for mono, 2 for stereo + + # Output format: 'url' or 'hex' (default is 'hex') + output_format = params.pop("output_format", "hex") + + request_body: Dict[str, Any] = { + "model": model, + "text": input, + "stream": False, # HTTP endpoint doesn't support streaming + "output_format": output_format, # 'url' or 'hex' + "voice_setting": { + "voice_id": voice_id, + "speed": speed, + "vol": vol, + "pitch": pitch, + }, + "audio_setting": { + "sample_rate": sample_rate, + "bitrate": bitrate, + "format": audio_format, + "channel": channel, + }, + } + + # Handle any remaining parameters from extra_body + extra_body = params.pop("extra_body", None) + if isinstance(extra_body, dict): + for key, value in extra_body.items(): + if value is not None and key not in request_body: + request_body[key] = value + + return TextToSpeechRequestData( + dict_body=request_body, + headers={"Content-Type": "application/json"}, + ) + + def transform_text_to_speech_response( + self, + model: str, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + ) -> "HttpxBinaryResponseContent": + """ + Transform MiniMax response to standard format. + + MiniMax returns JSON with base64-encoded audio data: + { + "base_resp": {"status_code": 0, "status_msg": "success"}, + "audio_file": "", + "extra_info": {...} + } + + We need to decode the base64 audio and return it as binary content. + """ + import base64 + import json + + from litellm.types.llms.openai import HttpxBinaryResponseContent + + try: + # Parse JSON response + response_json = raw_response.json() + + # MiniMax API response format check + # The API can return different structures: + # 1. {"data": {"audio": "..."}, "status": 0, ...} for HTTP endpoint + # 2. {"base_resp": {"status_code": 0, ...}, "audio_file": "..."} for older versions + + # Check for errors - MiniMax uses "status" field in HTTP endpoint response + # status: 0 = success, 2 = invalid api key, etc. + status = response_json.get("status") + if status is not None and status != 0: + ced = response_json.get("ced", "Unknown error") + error_detail = ced if ced else f"API returned status {status}" + raise MinimaxException( + status_code=raw_response.status_code, + message=f"MiniMax TTS error: {error_detail}", + headers=dict(raw_response.headers), + ) + + # Extract audio data + # MiniMax returns audio in "data" field + data = response_json.get("data", {}) + + # Check if response contains a URL (output_format='url') + audio_url = data.get("audio_url", None) + if audio_url: + # If URL format is used, we need to fetch the audio from the URL + # For now, return a response indicating URL mode (TODO: fetch audio from URL) + raise MinimaxException( + status_code=500, + message=f"URL output format is not yet supported. Use 'hex' format or fetch from URL: {audio_url}", + headers=dict(raw_response.headers), + ) + + # Get hex-encoded audio data + audio_hex = data.get("audio", "") or response_json.get("audio_file", "") + + if not audio_hex: + raise MinimaxException( + status_code=500, + message=f"No audio data in MiniMax response. Response keys: {list(response_json.keys())}", + headers=dict(raw_response.headers), + ) + + # MiniMax returns hex-encoded audio by default + # Try hex decoding first, fall back to base64 if that fails + try: + audio_bytes = bytes.fromhex(audio_hex) + except ValueError: + # If hex decoding fails, try base64 (for older API versions) + try: + audio_bytes = base64.b64decode(audio_hex) + except Exception as e: + raise MinimaxException( + status_code=500, + message=f"Failed to decode audio data: {str(e)}", + headers=dict(raw_response.headers), + ) + + # Create a new response with binary audio content + # We need to create a response that contains the decoded audio bytes + # Remove gzip encoding headers to avoid decompression issues + clean_headers = dict(raw_response.headers) + clean_headers.pop('content-encoding', None) + clean_headers.pop('transfer-encoding', None) + clean_headers['content-length'] = str(len(audio_bytes)) + + # Create a new response object with the binary content + binary_response = httpx.Response( + status_code=200, + headers=clean_headers, + content=audio_bytes, + request=raw_response.request, + ) + + return HttpxBinaryResponseContent(binary_response) + + except json.JSONDecodeError as e: + raise MinimaxException( + status_code=500, + message=f"Failed to parse MiniMax response: {str(e)}", + headers=dict(raw_response.headers), + ) + except Exception as e: + if isinstance(e, MinimaxException): + raise + raise MinimaxException( + status_code=500, + message=f"Error processing MiniMax response: {str(e)}", + headers=dict(raw_response.headers), + ) + + def get_complete_url( + self, + model: str, + api_base: Optional[str], + litellm_params: dict, + ) -> str: + """ + Construct the MiniMax endpoint URL. + """ + base_url = ( + api_base + or get_secret_str("MINIMAX_API_BASE") + or self.TTS_BASE_URL + ) + base_url = base_url.rstrip("/") + + # MiniMax uses a simple endpoint path + url = f"{base_url}{self.TTS_ENDPOINT_PATH}" + + return url + diff --git a/litellm/main.py b/litellm/main.py index fe2c2f333fc..6e069988726 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -110,17 +110,30 @@ from litellm.types.utils import ( RawRequestTypedDict, StreamingChoices, ) +from litellm.types.utils import ( + ModelResponseStream, + RawRequestTypedDict, + StreamingChoices, +) from litellm.utils import ( + Choices, Choices, CustomStreamWrapper, EmbeddingResponse, Message, ModelResponse, + EmbeddingResponse, + Message, + ModelResponse, ProviderConfigManager, TextChoices, TextCompletionResponse, TextCompletionStreamWrapper, TranscriptionResponse, + TextChoices, + TextCompletionResponse, + TextCompletionStreamWrapper, + TranscriptionResponse, Usage, _get_model_info_helper, add_provider_specific_params_to_optional_params, @@ -6507,6 +6520,46 @@ def speech( # noqa: PLR0915 api_key=api_key, **kwargs, ) + elif custom_llm_provider == "minimax": + from litellm.llms.minimax.text_to_speech.transformation import ( + MinimaxTextToSpeechConfig, + ) + + # MiniMax Text-to-Speech + if text_to_speech_provider_config is None: + text_to_speech_provider_config = MinimaxTextToSpeechConfig() + + minimax_config = cast( + MinimaxTextToSpeechConfig, text_to_speech_provider_config + ) + + if api_base is not None: + litellm_params_dict["api_base"] = api_base + if api_key is not None: + litellm_params_dict["api_key"] = api_key + + # Convert voice to string if it's a dict (minimax handler expects Optional[str]) + voice_str: Optional[str] = None + if isinstance(voice, str): + voice_str = voice + elif isinstance(voice, dict): + # Extract voice_id from dict if needed + voice_str = voice.get("voice_id") or voice.get("id") or voice.get("name") + + response = base_llm_http_handler.text_to_speech_handler( + model=model, + input=input, + voice=voice_str, + text_to_speech_provider_config=minimax_config, + text_to_speech_optional_params=optional_params, + custom_llm_provider=custom_llm_provider, + litellm_params=litellm_params_dict, + logging_obj=logging_obj, + timeout=timeout, + extra_headers=extra_headers, + client=client, + _is_async=aspeech or False, + ) elif custom_llm_provider == "aws_polly": from litellm.llms.aws_polly.text_to_speech.transformation import ( AWSPollyTextToSpeechConfig, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7f47ede07d1..513a4a554e0 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -19390,6 +19390,38 @@ "output_cost_per_token": 1.2e-06, "supports_system_messages": true }, + "minimax/speech-02-hd": { + "input_cost_per_character": 0.0001, + "litellm_provider": "minimax", + "mode": "audio_speech", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, + "minimax/speech-02-turbo": { + "input_cost_per_character": 0.00006, + "litellm_provider": "minimax", + "mode": "audio_speech", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, + "minimax/speech-2.6-hd": { + "input_cost_per_character": 0.0001, + "litellm_provider": "minimax", + "mode": "audio_speech", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, + "minimax/speech-2.6-turbo": { + "input_cost_per_character": 0.00006, + "litellm_provider": "minimax", + "mode": "audio_speech", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, "minimax/MiniMax-M2.1": { "input_cost_per_token": 3e-07, "output_cost_per_token": 1.2e-06, diff --git a/litellm/utils.py b/litellm/utils.py index a71b44facfc..102df5d595e 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8261,6 +8261,12 @@ class ProviderConfigManager: ) return VertexAITextToSpeechConfig() + elif litellm.LlmProviders.MINIMAX == provider: + from litellm.llms.minimax.text_to_speech.transformation import ( + MinimaxTextToSpeechConfig, + ) + + return MinimaxTextToSpeechConfig() elif litellm.LlmProviders.AWS_POLLY == provider: from litellm.llms.aws_polly.text_to_speech.transformation import ( AWSPollyTextToSpeechConfig, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7f47ede07d1..513a4a554e0 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -19390,6 +19390,38 @@ "output_cost_per_token": 1.2e-06, "supports_system_messages": true }, + "minimax/speech-02-hd": { + "input_cost_per_character": 0.0001, + "litellm_provider": "minimax", + "mode": "audio_speech", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, + "minimax/speech-02-turbo": { + "input_cost_per_character": 0.00006, + "litellm_provider": "minimax", + "mode": "audio_speech", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, + "minimax/speech-2.6-hd": { + "input_cost_per_character": 0.0001, + "litellm_provider": "minimax", + "mode": "audio_speech", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, + "minimax/speech-2.6-turbo": { + "input_cost_per_character": 0.00006, + "litellm_provider": "minimax", + "mode": "audio_speech", + "supported_endpoints": [ + "/v1/audio/speech" + ] + }, "minimax/MiniMax-M2.1": { "input_cost_per_token": 3e-07, "output_cost_per_token": 1.2e-06, diff --git a/tests/llm_translation/test_minimax_tts.py b/tests/llm_translation/test_minimax_tts.py new file mode 100644 index 00000000000..88ddf9be0b1 --- /dev/null +++ b/tests/llm_translation/test_minimax_tts.py @@ -0,0 +1,371 @@ +""" +Tests for MiniMax Text-to-Speech integration +""" + +import os +import sys +from pathlib import Path +from unittest.mock import MagicMock, Mock, patch + +import pytest + +sys.path.insert( + 0, os.path.abspath("../..") +) # Adds the parent directory to the system path + +import litellm +from litellm import speech +from litellm.llms.minimax.text_to_speech.transformation import ( + MinimaxTextToSpeechConfig, +) + + +class TestMinimaxTextToSpeechConfig: + """Test MiniMax TTS configuration and parameter mapping""" + + def test_get_supported_openai_params(self): + """Test that supported OpenAI params are correctly defined""" + config = MinimaxTextToSpeechConfig() + supported_params = config.get_supported_openai_params("speech-2.6-hd") + + assert "voice" in supported_params + assert "response_format" in supported_params + assert "speed" in supported_params + + def test_voice_mapping(self): + """Test OpenAI voice to MiniMax voice_id mapping""" + config = MinimaxTextToSpeechConfig() + + # Test OpenAI voice mappings + assert config._extract_voice_id("alloy") == "male-qn-qingse" + assert config._extract_voice_id("echo") == "male-qn-jingying" + assert config._extract_voice_id("nova") == "female-yujie" + + # Test custom voice passthrough + assert config._extract_voice_id("custom-voice-id") == "custom-voice-id" + + def test_format_mapping(self): + """Test response format mapping""" + config = MinimaxTextToSpeechConfig() + + assert config.FORMAT_MAPPINGS["mp3"] == "mp3" + assert config.FORMAT_MAPPINGS["pcm"] == "pcm" + assert config.FORMAT_MAPPINGS["wav"] == "wav" + assert config.FORMAT_MAPPINGS["flac"] == "flac" + + def test_map_openai_params_basic(self): + """Test basic parameter mapping from OpenAI to MiniMax format""" + config = MinimaxTextToSpeechConfig() + + optional_params = { + "response_format": "mp3", + "speed": 1.5, + } + + voice, mapped_params = config.map_openai_params( + model="speech-2.6-hd", + optional_params=optional_params, + voice="alloy", + ) + + assert voice == "male-qn-qingse" + assert mapped_params["format"] == "mp3" + assert mapped_params["speed"] == 1.5 + assert mapped_params["voice_id"] == "male-qn-qingse" + + def test_map_openai_params_speed_clamping(self): + """Test that speed is clamped to MiniMax's supported range""" + config = MinimaxTextToSpeechConfig() + + # Test speed too high + optional_params = {"speed": 5.0} + _, mapped_params = config.map_openai_params( + model="speech-2.6-hd", + optional_params=optional_params, + voice="alloy", + ) + assert mapped_params["speed"] == 2.0 # Clamped to max + + # Test speed too low + optional_params = {"speed": 0.1} + _, mapped_params = config.map_openai_params( + model="speech-2.6-hd", + optional_params=optional_params, + voice="alloy", + ) + assert mapped_params["speed"] == 0.5 # Clamped to min + + def test_map_openai_params_with_extra_body(self): + """Test that extra_body parameters are passed through""" + config = MinimaxTextToSpeechConfig() + + optional_params = { + "extra_body": { + "vol": 1.5, + "pitch": 2, + "sample_rate": 24000, + } + } + + _, mapped_params = config.map_openai_params( + model="speech-2.6-hd", + optional_params=optional_params, + voice="alloy", + ) + + assert mapped_params["vol"] == 1.5 + assert mapped_params["pitch"] == 2 + assert mapped_params["sample_rate"] == 24000 + + def test_validate_environment_with_api_key(self): + """Test environment validation with API key""" + config = MinimaxTextToSpeechConfig() + headers = {} + + result_headers = config.validate_environment( + headers=headers, + model="speech-2.6-hd", + api_key="test-api-key", + ) + + assert "Authorization" in result_headers + assert result_headers["Authorization"] == "Bearer test-api-key" + assert result_headers["Content-Type"] == "application/json" + + def test_validate_environment_missing_api_key(self): + """Test that validation fails without API key""" + config = MinimaxTextToSpeechConfig() + headers = {} + + # Mock both litellm.api_key and get_secret_str to return None + import litellm + from unittest.mock import patch + + original_api_key = litellm.api_key + try: + litellm.api_key = None + with patch("litellm.llms.minimax.text_to_speech.transformation.get_secret_str", return_value=None): + with pytest.raises(ValueError, match="MiniMax API key is required"): + config.validate_environment( + headers=headers, + model="speech-2.6-hd", + api_key=None, + ) + finally: + litellm.api_key = original_api_key + + def test_transform_text_to_speech_request(self): + """Test request transformation to MiniMax format""" + config = MinimaxTextToSpeechConfig() + + optional_params = { + "voice_id": "male-qn-qingse", + "speed": 1.2, + "format": "mp3", + "vol": 1.0, + "pitch": 0, + "sample_rate": 32000, + "bitrate": 128000, + "channel": 1, + } + + result = config.transform_text_to_speech_request( + model="speech-2.6-hd", + input="Hello, world!", + voice="male-qn-qingse", + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "dict_body" in result + body = result["dict_body"] + + assert body["model"] == "speech-2.6-hd" + assert body["text"] == "Hello, world!" + assert body["stream"] is False + assert body["voice_setting"]["voice_id"] == "male-qn-qingse" + assert body["voice_setting"]["speed"] == 1.2 + assert body["audio_setting"]["format"] == "mp3" + assert body["audio_setting"]["sample_rate"] == 32000 + + def test_get_complete_url(self): + """Test URL construction""" + config = MinimaxTextToSpeechConfig() + + url = config.get_complete_url( + model="speech-2.6-hd", + api_base=None, + litellm_params={}, + ) + + assert url == "https://api.minimax.io/v1/t2a_v2" + + def test_get_complete_url_custom_base(self): + """Test URL construction with custom API base""" + config = MinimaxTextToSpeechConfig() + + url = config.get_complete_url( + model="speech-2.6-hd", + api_base="https://custom.api.com", + litellm_params={}, + ) + + assert url == "https://custom.api.com/v1/t2a_v2" + + +class TestMinimaxSpeechIntegration: + """Integration tests for MiniMax TTS via litellm.speech()""" + + @pytest.mark.skip(reason="Requires MiniMax API key") + def test_speech_basic(self): + """Test basic speech synthesis call""" + # This test requires a real API key + os.environ["MINIMAX_API_KEY"] = "your-api-key-here" + + speech_file_path = Path(__file__).parent / "test_minimax_speech.mp3" + + response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="Hello, this is a test of MiniMax text to speech.", + ) + + response.stream_to_file(speech_file_path) + + # Verify file was created + assert speech_file_path.exists() + assert speech_file_path.stat().st_size > 0 + + # Clean up + speech_file_path.unlink() + + @pytest.mark.skip(reason="Requires MiniMax API key") + def test_speech_with_custom_params(self): + """Test speech synthesis with custom parameters""" + os.environ["MINIMAX_API_KEY"] = "your-api-key-here" + + speech_file_path = Path(__file__).parent / "test_minimax_speech_custom.mp3" + + response = speech( + model="minimax/speech-2.6-turbo", + voice="nova", + input="Testing custom parameters.", + speed=1.5, + response_format="mp3", + extra_body={ + "vol": 1.2, + "pitch": 1, + "sample_rate": 24000, + }, + ) + + response.stream_to_file(speech_file_path) + + # Verify file was created + assert speech_file_path.exists() + assert speech_file_path.stat().st_size > 0 + + # Clean up + speech_file_path.unlink() + + def test_speech_mock_response(self): + """Test speech synthesis with mocked response""" + from unittest.mock import MagicMock, patch + + # Create mock audio data (hex-encoded as MiniMax returns) + mock_audio_bytes = b"fake audio data for testing" + mock_audio_hex = mock_audio_bytes.hex() + + mock_response_json = { + "data": { + "audio": mock_audio_hex, + "status": 0, + "ced": "" + }, + "extra_info": {}, + } + + with patch("litellm.llms.custom_httpx.llm_http_handler.BaseLLMHTTPHandler.text_to_speech_handler") as mock_tts: + # Create a mock httpx.Response + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.headers = {} + mock_response.json.return_value = mock_response_json + mock_response.content = mock_audio_bytes + + # Mock the response wrapper + from litellm.types.llms.openai import HttpxBinaryResponseContent + mock_binary_response = HttpxBinaryResponseContent(mock_response) + mock_tts.return_value = mock_binary_response + + # This would normally make a real API call + # but we're mocking it for testing + response = speech( + model="minimax/speech-2.6-hd", + voice="alloy", + input="Test input", + api_key="test-key", + ) + + # Verify the mock was called + assert mock_tts.called + + +class TestMinimaxProviderRegistration: + """Test that MiniMax is properly registered as a provider""" + + def test_minimax_in_llm_providers(self): + """Test that MINIMAX is in LlmProviders enum""" + from litellm.types.utils import LlmProviders + + assert hasattr(LlmProviders, "MINIMAX") + assert LlmProviders.MINIMAX.value == "minimax" + + def test_minimax_in_provider_list(self): + """Test that minimax is in the provider list""" + assert litellm.LlmProviders.MINIMAX in litellm.provider_list + + def test_get_provider_text_to_speech_config(self): + """Test that MiniMax TTS config can be retrieved""" + from litellm.utils import ProviderConfigManager + + config = ProviderConfigManager.get_provider_text_to_speech_config( + model="speech-2.6-hd", + provider=litellm.LlmProviders.MINIMAX, + ) + + assert config is not None + assert isinstance(config, MinimaxTextToSpeechConfig) + + def test_get_llm_provider_minimax(self): + """Test that get_llm_provider correctly identifies MiniMax models""" + from litellm import get_llm_provider + + model, provider, api_key, api_base = get_llm_provider( + model="minimax/speech-2.6-hd" + ) + + assert model == "speech-2.6-hd" + assert provider == "minimax" + + +if __name__ == "__main__": + # Run basic tests + test_config = TestMinimaxTextToSpeechConfig() + test_config.test_get_supported_openai_params() + test_config.test_voice_mapping() + test_config.test_format_mapping() + test_config.test_map_openai_params_basic() + test_config.test_map_openai_params_speed_clamping() + test_config.test_transform_text_to_speech_request() + test_config.test_get_complete_url() + + test_registration = TestMinimaxProviderRegistration() + test_registration.test_minimax_in_llm_providers() + test_registration.test_minimax_in_provider_list() + test_registration.test_get_provider_text_to_speech_config() + test_registration.test_get_llm_provider_minimax() + + print("All basic tests passed!") +