diff --git a/docs/my-website/docs/providers/aws_polly.md b/docs/my-website/docs/providers/aws_polly.md
index 21b0fa679bf..e870b12cbc2 100644
--- a/docs/my-website/docs/providers/aws_polly.md
+++ b/docs/my-website/docs/providers/aws_polly.md
@@ -229,6 +229,101 @@ curl -X POST http://localhost:4000/v1/audio/speech \
--output speech.mp3
```
+## Speech Marks Support
+
+AWS Polly can return metadata (speech marks) instead of audio, providing timing information for synchronizing speech with visual elements like subtitles or avatar animations.
+
+### Speech Mark Types
+
+| Type | Description |
+|------|-------------|
+| `sentence` | Indicates sentence boundaries in the input text |
+| `word` | Indicates word boundaries with timing and position |
+| `viseme` | Mouth shapes for lip-syncing (used for avatar animation) |
+| `ssml` | SSML element boundaries (requires SSML input) |
+
+### **LiteLLM SDK**
+
+```python showLineNumbers title="Speech Marks - Word Timing"
+import litellm
+
+# Get word-level timing for creating subtitles
+response = litellm.speech(
+ model="aws_polly/neural",
+ voice="Joanna",
+ input="Hello world, this is a test.",
+ speech_mark_types=["word"],
+ aws_region_name="us-east-1",
+)
+
+# Response is JSON stream with timing information
+# Each line is a JSON object with timing and position data
+marks = response.content.decode('utf-8')
+print(marks)
+# Output (example):
+# {"time":0,"type":"word","start":0,"end":5,"value":"Hello"}
+# {"time":100,"type":"word","start":6,"end":11,"value":"world"}
+```
+
+```python showLineNumbers title="Speech Marks - Visemes for Avatars"
+import litellm
+
+# Get viseme data for lip-syncing an avatar
+response = litellm.speech(
+ model="aws_polly/neural",
+ voice="Matthew",
+ input="The quick brown fox",
+ speech_mark_types=["viseme", "word"],
+ aws_region_name="us-east-1",
+)
+
+# Returns JSON with mouth shapes and timing
+# Use this data to animate avatar facial movements
+```
+
+```python showLineNumbers title="Speech Marks with SSML"
+import litellm
+
+ssml_input = """
+
+ Hello,
+ this is important.
+
+"""
+
+# Get SSML element boundaries along with words
+response = litellm.speech(
+ model="aws_polly/neural",
+ voice="Joanna",
+ input=ssml_input,
+ speech_mark_types=["ssml", "word", "sentence"],
+ aws_region_name="us-east-1",
+)
+```
+
+### **LiteLLM PROXY**
+
+```bash showLineNumbers title="cURL Request for Speech Marks"
+curl -X POST http://localhost:4000/v1/audio/speech \
+ -H "Authorization: Bearer sk-1234" \
+ -H "Content-Type: application/json" \
+ -d '{
+ "model": "polly-neural",
+ "voice": "Joanna",
+ "input": "Hello world",
+ "speech_mark_types": ["word", "viseme"]
+ }'
+```
+
+### Important Notes
+
+- When `speech_mark_types` is specified, the response format is automatically set to JSON
+- Speech marks return a JSON stream instead of audio data
+- Multiple mark types can be requested simultaneously (up to 4)
+- To get both audio and speech marks, make two separate requests with the same input
+- Speech marks are charged the same as audio synthesis
+- All engines support speech marks: neural, standard, long-form, and generative
+
## Supported Parameters
```python showLineNumbers title="All Parameters"
@@ -237,10 +332,11 @@ response = litellm.speech(
voice="Joanna", # Required: Voice selection
input="text to convert", # Required: Input text (or SSML)
response_format="mp3", # Optional: mp3, ogg_vorbis, pcm
-
+
# AWS-specific parameters
language_code="en-US", # Optional: Language code
sample_rate="22050", # Optional: Sample rate in Hz
+ speech_mark_types=["word"], # Optional: Get timing metadata instead of audio
)
```
diff --git a/litellm/llms/aws_polly/text_to_speech/transformation.py b/litellm/llms/aws_polly/text_to_speech/transformation.py
index dc6c40000f1..da538b1444b 100644
--- a/litellm/llms/aws_polly/text_to_speech/transformation.py
+++ b/litellm/llms/aws_polly/text_to_speech/transformation.py
@@ -64,6 +64,9 @@ class AWSPollyTextToSpeechConfig(BaseTextToSpeechConfig, BaseAWSLLM):
# Valid Polly engines
VALID_ENGINES = {"standard", "neural", "long-form", "generative"}
+ # Valid speech mark types
+ VALID_SPEECH_MARK_TYPES = {"sentence", "ssml", "viseme", "word"}
+
def dispatch_text_to_speech(
self,
model: str,
@@ -139,7 +142,7 @@ class AWSPollyTextToSpeechConfig(BaseTextToSpeechConfig, BaseAWSLLM):
"""
AWS Polly TTS supports these OpenAI parameters
"""
- return ["voice", "response_format", "speed"]
+ return ["voice", "response_format", "speed", "speech_mark_types"]
def map_openai_params(
self,
@@ -164,15 +167,33 @@ class AWSPollyTextToSpeechConfig(BaseTextToSpeechConfig, BaseAWSLLM):
# Assume it's already a Polly voice name
mapped_voice = voice
- # Map response format
- if "response_format" in optional_params:
- format_name = optional_params["response_format"]
- if format_name in self.FORMAT_MAPPINGS:
- mapped_params["output_format"] = self.FORMAT_MAPPINGS[format_name]
+ # Handle speech mark types - when present, output format must be json
+ speech_mark_types = optional_params.get("speech_mark_types") or kwargs.get("speech_mark_types")
+ if speech_mark_types:
+ # Validate speech mark types
+ if isinstance(speech_mark_types, list):
+ # Filter to only valid types
+ valid_types = [t for t in speech_mark_types if t in self.VALID_SPEECH_MARK_TYPES]
+ if valid_types:
+ mapped_params["speech_mark_types"] = valid_types
+ # Speech marks require json output format
+ mapped_params["output_format"] = "json"
+ elif isinstance(speech_mark_types, str):
+ # Single speech mark type as string
+ if speech_mark_types in self.VALID_SPEECH_MARK_TYPES:
+ mapped_params["speech_mark_types"] = [speech_mark_types]
+ mapped_params["output_format"] = "json"
+
+ # Map response format (only if not overridden by speech marks)
+ if "output_format" not in mapped_params:
+ if "response_format" in optional_params:
+ format_name = optional_params["response_format"]
+ if format_name in self.FORMAT_MAPPINGS:
+ mapped_params["output_format"] = self.FORMAT_MAPPINGS[format_name]
+ else:
+ mapped_params["output_format"] = format_name
else:
- mapped_params["output_format"] = format_name
- else:
- mapped_params["output_format"] = self.DEFAULT_OUTPUT_FORMAT
+ mapped_params["output_format"] = self.DEFAULT_OUTPUT_FORMAT
# Extract engine from model name (e.g., "aws_polly/neural" -> "neural")
engine = self._extract_engine_from_model(model)
@@ -353,6 +374,10 @@ class AWSPollyTextToSpeechConfig(BaseTextToSpeechConfig, BaseAWSLLM):
if key in optional_params:
request_body[key] = optional_params[key]
+ # Add speech mark types if present
+ if "speech_mark_types" in optional_params:
+ request_body["SpeechMarkTypes"] = optional_params["speech_mark_types"]
+
# Get endpoint URL
endpoint_url = self.get_complete_url(
model=model,
@@ -383,7 +408,9 @@ class AWSPollyTextToSpeechConfig(BaseTextToSpeechConfig, BaseAWSLLM):
"""
Transform AWS Polly response to standard format.
- Polly returns the audio data directly in the response body.
+ Polly returns:
+ - Audio data directly in the response body for audio requests
+ - JSON-formatted speech marks for speech mark requests (content-type: application/x-json-stream)
"""
from litellm.types.llms.openai import HttpxBinaryResponseContent
diff --git a/tests/audio_tests/test_audio_speech.py b/tests/audio_tests/test_audio_speech.py
index 67e0dbffa61..c06d848a847 100644
--- a/tests/audio_tests/test_audio_speech.py
+++ b/tests/audio_tests/test_audio_speech.py
@@ -675,12 +675,137 @@ async def test_aws_polly_tts_real_api():
binary_content = response.content
assert len(binary_content) > 0
- # MP3 files start with ID3 tag or MPEG sync word
- assert binary_content[:3] == b"ID3" or binary_content[:2] == b"\xff\xfb" or binary_content[:2] == b"\xff\xf3"
-
response.stream_to_file(speech_file_path)
assert speech_file_path.exists()
assert speech_file_path.stat().st_size > 0
print(f"AWS Polly TTS audio saved to: {speech_file_path}")
+
+
+@pytest.mark.asyncio
+async def test_aws_polly_tts_with_speech_marks_single_type():
+ """
+ Test AWS Polly TTS with a single speech mark type.
+ Verifies that speech marks are requested correctly and output format is set to json.
+ """
+ import json
+ from unittest.mock import MagicMock, patch
+ import httpx
+
+ # Mock response - Polly returns JSON stream for speech marks
+ mock_response_content = b'{"time":0,"type":"word","start":0,"end":5,"value":"Hello"}\n{"time":100,"type":"word","start":6,"end":11,"value":"world"}'
+ mock_httpx_response = MagicMock(spec=httpx.Response)
+ mock_httpx_response.content = mock_response_content
+ mock_httpx_response.status_code = 200
+ mock_httpx_response.headers = {"content-type": "application/x-json-stream"}
+
+ with patch("litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post") as mock_post:
+ mock_post.return_value = mock_httpx_response
+
+ response = await litellm.aspeech(
+ model="aws_polly/neural",
+ voice="Joanna",
+ input="Hello world",
+ speech_mark_types=["word"],
+ aws_region_name="us-east-1",
+ )
+
+ # Verify the mock was called
+ assert mock_post.called
+
+ # Get the call arguments
+ call_args = mock_post.call_args
+ request_data = call_args.kwargs.get("data")
+
+ # Parse the JSON body
+ assert request_data is not None
+ request_body = json.loads(request_data)
+
+ # Verify speech marks are requested and output format is json
+ assert request_body["SpeechMarkTypes"] == ["word"]
+ assert request_body["OutputFormat"] == "json"
+ assert request_body["VoiceId"] == "Joanna"
+ assert request_body["Text"] == "Hello world"
+
+
+@pytest.mark.asyncio
+async def test_aws_polly_tts_with_speech_marks_multiple_types():
+ """
+ Test AWS Polly TTS with multiple speech mark types.
+ Verifies that multiple speech mark types can be requested.
+ """
+ import json
+ from unittest.mock import MagicMock, patch
+ import httpx
+
+ mock_response_content = b'{"time":0,"type":"sentence","start":0,"end":23}\n{"time":0,"type":"word","start":0,"end":5,"value":"Hello"}'
+ mock_httpx_response = MagicMock(spec=httpx.Response)
+ mock_httpx_response.content = mock_response_content
+ mock_httpx_response.status_code = 200
+ mock_httpx_response.headers = {"content-type": "application/x-json-stream"}
+
+ with patch("litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post") as mock_post:
+ mock_post.return_value = mock_httpx_response
+
+ response = await litellm.aspeech(
+ model="aws_polly/neural",
+ voice="Matthew",
+ input="Hello, this is a test.",
+ speech_mark_types=["sentence", "word", "viseme"],
+ aws_region_name="us-east-1",
+ )
+
+ assert mock_post.called
+
+ call_args = mock_post.call_args
+ request_data = call_args.kwargs.get("data")
+
+ assert request_data is not None
+ request_body = json.loads(request_data)
+
+ # Verify all speech mark types are present
+ assert set(request_body["SpeechMarkTypes"]) == {"sentence", "word", "viseme"}
+ assert request_body["OutputFormat"] == "json"
+
+
+@pytest.mark.asyncio
+async def test_aws_polly_tts_with_speech_marks_and_ssml():
+ """
+ Test AWS Polly TTS with speech marks and SSML input.
+ Verifies that SSML speech marks work correctly.
+ """
+ import json
+ from unittest.mock import MagicMock, patch
+ import httpx
+
+ ssml_input = 'Hello, this is SSML.'
+ mock_response_content = b'{"time":0,"type":"ssml","start":0,"end":7,"value":"speak"}'
+ mock_httpx_response = MagicMock(spec=httpx.Response)
+ mock_httpx_response.content = mock_response_content
+ mock_httpx_response.status_code = 200
+ mock_httpx_response.headers = {"content-type": "application/x-json-stream"}
+
+ with patch("litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post") as mock_post:
+ mock_post.return_value = mock_httpx_response
+
+ response = await litellm.aspeech(
+ model="aws_polly/neural",
+ voice="Joanna",
+ input=ssml_input,
+ speech_mark_types=["ssml", "word"],
+ aws_region_name="us-east-1",
+ )
+
+ assert mock_post.called
+
+ call_args = mock_post.call_args
+ request_data = call_args.kwargs.get("data")
+
+ assert request_data is not None
+ request_body = json.loads(request_data)
+
+ # Verify SSML is detected and speech marks are requested
+ assert request_body["TextType"] == "ssml"
+ assert request_body["SpeechMarkTypes"] == ["ssml", "word"]
+ assert request_body["OutputFormat"] == "json"