[Feat] Add Eleven Labs - Speech To Text Support on LiteLLM (#12119)

* add ELEVENLABS as a provider

* add deepgram to main.py

* add ElevenLabsException

* add ElevenLabsAudioTranscriptionConfig

* add transform_audio_transcription_response

* TestElevenLabsAudioTranscription

* add elevenlabs/scribe_v1 to model cost map

* add ElevenLabsAudioTranscriptionConfig

* add AudioTranscriptionRequestData

* add ElevenLabs transform

* use AudioTranscriptionRequestData

* refactoring fixes

* add ProcessedAudioFile util for reading audio files

* test_elevenlabs_diarize_parameter_passthrough

* docs eleven labs

* docs fixes

* fix code qa checks

* fixes - audio transcription

* ui - add ElevenLabs logo

* add elevenlabs logo

* docs - ElevenLabs

* test fix elevenlabs
This commit is contained in:
Ishaan Jaff 2025-06-27 17:50:49 -07:00 • committed by GitHub
parent 041db0268c
commit ebf6395bc1
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
24 changed files with 1109 additions and 131 deletions

View file

@ -0,0 +1,231 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
# ElevenLabs
ElevenLabs provides high-quality AI voice technology, including speech-to-text capabilities through their transcription API.
| Property | Details |
|----------|---------|
| Description | ElevenLabs offers advanced AI voice technology with speech-to-text transcription capabilities that support multiple languages and speaker diarization. |
| Provider Route on LiteLLM | `elevenlabs/` |
| Provider Doc | [ElevenLabs API ↗](https://elevenlabs.io/docs/api-reference) |
| Supported Endpoints | `/audio/transcriptions` |
## Quick Start
### LiteLLM Python SDK
<Tabs>
<TabItem value="basic" label="Basic Usage">
```python showLineNumbers title="Basic audio transcription with ElevenLabs"
import litellm
# Transcribe audio file
with open("audio.mp3", "rb") as audio_file:
response = litellm.transcription(
model="elevenlabs/scribe_v1",
file=audio_file,
api_key="your-elevenlabs-api-key" # or set ELEVENLABS_API_KEY env var
)
print(response.text)
```
</TabItem>
<TabItem value="advanced" label="Advanced Features">
```python showLineNumbers title="Audio transcription with advanced features"
import litellm
# Transcribe with speaker diarization and language specification
with open("audio.wav", "rb") as audio_file:
response = litellm.transcription(
model="elevenlabs/scribe_v1",
file=audio_file,
language="en", # Language hint (maps to language_code)
temperature=0.3, # Control randomness in transcription
diarize=True, # Enable speaker diarization
api_key="your-elevenlabs-api-key"
)
print(f"Transcription: {response.text}")
print(f"Language: {response.language}")
# Access word-level timestamps if available
if hasattr(response, 'words') and response.words:
for word_info in response.words:
print(f"Word: {word_info['word']}, Start: {word_info['start']}, End: {word_info['end']}")
```
</TabItem>
<TabItem value="async" label="Async Usage">
```python showLineNumbers title="Async audio transcription"
import litellm
import asyncio
async def transcribe_audio():
with open("audio.mp3", "rb") as audio_file:
response = await litellm.atranscription(
model="elevenlabs/scribe_v1",
file=audio_file,
api_key="your-elevenlabs-api-key"
)
return response.text
# Run async transcription
result = asyncio.run(transcribe_audio())
print(result)
```
</TabItem>
</Tabs>
### LiteLLM Proxy
#### 1. Configure your proxy
<Tabs>
<TabItem value="config-yaml" label="config.yaml">
```yaml showLineNumbers title="ElevenLabs configuration in config.yaml"
model_list:
- model_name: elevenlabs-transcription
litellm_params:
model: elevenlabs/scribe_v1
api_key: os.environ/ELEVENLABS_API_KEY
general_settings:
master_key: your-master-key
```
</TabItem>
<TabItem value="env-vars" label="Environment Variables">
```bash showLineNumbers title="Required environment variables"
export ELEVENLABS_API_KEY="your-elevenlabs-api-key"
export LITELLM_MASTER_KEY="your-master-key"
```
</TabItem>
</Tabs>
#### 2. Start the proxy
```bash showLineNumbers title="Start LiteLLM proxy server"
litellm --config config.yaml
# Proxy will be available at http://localhost:4000
```
#### 3. Make transcription requests
<Tabs>
<TabItem value="curl" label="Curl">
```bash showLineNumbers title="Audio transcription with curl"
curl http://localhost:4000/v1/audio/transcriptions \
-H "Authorization: Bearer $LITELLM_API_KEY" \
-H "Content-Type: multipart/form-data" \
-F file="@audio.mp3" \
-F model="elevenlabs-transcription" \
-F language="en" \
-F temperature="0.3"
```
</TabItem>
<TabItem value="openai-sdk" label="OpenAI Python SDK">
```python showLineNumbers title="Using OpenAI SDK with LiteLLM proxy"
from openai import OpenAI
# Initialize client with your LiteLLM proxy URL
client = OpenAI(
base_url="http://localhost:4000",
api_key="your-litellm-api-key"
)
# Transcribe audio file
with open("audio.mp3", "rb") as audio_file:
response = client.audio.transcriptions.create(
model="elevenlabs-transcription",
file=audio_file,
language="en",
temperature=0.3,
# ElevenLabs-specific parameters
diarize=True,
speaker_boost=True,
custom_vocabulary="technical,AI,machine learning"
)
print(response.text)
```
</TabItem>
<TabItem value="javascript" label="JavaScript/Node.js">
```javascript showLineNumbers title="Audio transcription with JavaScript"
import OpenAI from 'openai';
import fs from 'fs';
const openai = new OpenAI({
baseURL: 'http://localhost:4000',
apiKey: 'your-litellm-api-key'
});
async function transcribeAudio() {
const response = await openai.audio.transcriptions.create({
file: fs.createReadStream('audio.mp3'),
model: 'elevenlabs-transcription',
language: 'en',
temperature: 0.3,
diarize: true,
speaker_boost: true
});
console.log(response.text);
}
transcribeAudio();
```
</TabItem>
</Tabs>
## Response Format
ElevenLabs returns transcription responses in OpenAI-compatible format:
```json showLineNumbers title="Example transcription response"
{
"text": "Hello, this is a sample transcription with multiple speakers.",
"task": "transcribe",
"language": "en",
"words": [
{
"word": "Hello",
"start": 0.0,
"end": 0.5
},
{
"word": "this",
"start": 0.5,
"end": 0.8
}
]
}
```
### Common Issues
1. **Invalid API Key**: Ensure `ELEVENLABS_API_KEY` is set correctly

View file

@ -415,6 +415,7 @@ const sidebars = {
"providers/groq",
"providers/github",
"providers/deepseek",
"providers/elevenlabs",
"providers/fireworks_ai",
"providers/clarifai",
"providers/vllm",

View file

@ -478,6 +478,7 @@ nscale_models: List = []
nebius_models: List = []
nebius_embedding_models: List = []
deepgram_models: List = []
elevenlabs_models: List = []
def is_bedrock_pricing_only_model(key: str) -> bool:
@ -651,6 +652,8 @@ def add_known_models():
featherless_ai_models.append(key)
elif value.get("litellm_provider") == "deepgram":
deepgram_models.append(key)
elif value.get("litellm_provider") == "elevenlabs":
elevenlabs_models.append(key)
add_known_models()
@ -733,6 +736,7 @@ model_list = (
+ featherless_ai_models
+ nscale_models
+ deepgram_models
+ elevenlabs_models
)
model_list_set = set(model_list)
@ -797,6 +801,7 @@ models_by_provider: dict = {
"nscale": nscale_models,
"featherless_ai": featherless_ai_models,
"deepgram": deepgram_models,
"elevenlabs": elevenlabs_models,
}
# mapping for those models which have larger equivalents

View file

@ -3,10 +3,110 @@ Utils used for litellm.transcription() and litellm.atranscription()
"""
import os
from dataclasses import dataclass
from litellm.types.files import get_file_mime_type_from_extension
from litellm.types.utils import FileTypes
@dataclass
class ProcessedAudioFile:
"""
Processed audio file data.
Attributes:
file_content: The binary content of the audio file
filename: The filename (extracted or generated)
content_type: The MIME type of the audio file
"""
file_content: bytes
filename: str
content_type: str
def process_audio_file(audio_file: FileTypes) -> ProcessedAudioFile:
"""
Common utility function to process audio files for audio transcription APIs.
Handles various input types:
- File paths (str, os.PathLike)
- Raw bytes/bytearray
- Tuples (filename, content, optional content_type)
- File-like objects with read() method
Args:
audio_file: The audio file input in various formats
Returns:
ProcessedAudioFile: Structured data with file content, filename, and content type
Raises:
ValueError: If audio_file type is unsupported or content cannot be extracted
"""
file_content = None
filename = None
if isinstance(audio_file, (bytes, bytearray)):
# Raw bytes
filename = 'audio.wav'
file_content = bytes(audio_file)
elif isinstance(audio_file, (str, os.PathLike)):
# File path or PathLike
file_path = str(audio_file)
with open(file_path, 'rb') as f:
file_content = f.read()
filename = file_path.split('/')[-1]
elif isinstance(audio_file, tuple):
# Tuple format: (filename, content, content_type) or (filename, content)
if len(audio_file) >= 2:
filename = audio_file[0] or 'audio.wav'
content = audio_file[1]
if isinstance(content, (bytes, bytearray)):
file_content = bytes(content)
elif isinstance(content, (str, os.PathLike)):
# File path or PathLike
with open(str(content), 'rb') as f:
file_content = f.read()
elif hasattr(content, 'read'):
# File-like object
file_content = content.read()
if hasattr(content, 'seek'):
content.seek(0)
else:
raise ValueError(f"Unsupported content type in tuple: {type(content)}")
else:
raise ValueError("Tuple must have at least 2 elements: (filename, content)")
elif hasattr(audio_file, 'read') and not isinstance(audio_file, (str, bytes, bytearray, tuple, os.PathLike)):
# File-like object (IO) - check this after all other types
filename = getattr(audio_file, 'name', 'audio.wav')
file_content = audio_file.read() # type: ignore
# Reset file pointer if possible
if hasattr(audio_file, 'seek'):
audio_file.seek(0) # type: ignore
else:
raise ValueError(f"Unsupported audio_file type: {type(audio_file)}")
if file_content is None:
raise ValueError("Could not extract file content from audio_file")
# Determine content type using LiteLLM's file type utilities
content_type = 'audio/wav' # Default fallback
if filename:
try:
# Extract extension from filename
extension = filename.split('.')[-1].lower() if '.' in filename else 'wav'
content_type = get_file_mime_type_from_extension(extension)
except ValueError:
# If extension is not recognized, fallback to audio/wav
content_type = 'audio/wav'
return ProcessedAudioFile(
file_content=file_content,
filename=filename,
content_type=content_type
)
def get_audio_file_name(file_obj: FileTypes) -> str:
"""
Safely get the name of a file-like object or return its string representation.

View file

@ -252,6 +252,16 @@ def get_supported_openai_params( # noqa: PLR0915
model=model
)
)
elif custom_llm_provider == "elevenlabs":
if request_type == "transcription":
from litellm.llms.elevenlabs.audio_transcription.transformation import (
ElevenLabsAudioTranscriptionConfig,
)
return (
ElevenLabsAudioTranscriptionConfig().get_supported_openai_params(
model=model
)
)
elif custom_llm_provider in litellm._custom_providers:
if request_type == "chat_completion":
provider_config = litellm.ProviderConfigManager.get_provider_chat_config(

View file

@ -1,5 +1,6 @@
from abc import ABC, abstractmethod
from typing import TYPE_CHECKING, Any, List, Optional, Union
from dataclasses import dataclass
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union
import httpx
@ -8,7 +9,7 @@ from litellm.types.llms.openai import (
AllMessageValues,
OpenAIAudioTranscriptionOptionalParams,
)
from litellm.types.utils import FileTypes, ModelResponse
from litellm.types.utils import FileTypes, ModelResponse, TranscriptionResponse
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
@ -18,6 +19,21 @@ else:
LiteLLMLoggingObj = Any
@dataclass
class AudioTranscriptionRequestData:
"""
Structured data for audio transcription requests.
Attributes:
data: The request data (form data for multipart, json data for regular requests)
files: Optional files dict for multipart form data
content_type: Optional content type override
"""
data: Union[dict, bytes]
files: Optional[dict] = None
content_type: Optional[str] = None
class BaseAudioTranscriptionConfig(BaseConfig, ABC):
@abstractmethod
def get_supported_openai_params(
@ -50,11 +66,21 @@ class BaseAudioTranscriptionConfig(BaseConfig, ABC):
audio_file: FileTypes,
optional_params: dict,
litellm_params: dict,
) -> Union[dict, bytes]:
) -> Union[AudioTranscriptionRequestData, Dict]:
raise NotImplementedError(
"AudioTranscriptionConfig needs a request transformation for audio transcription models"
)
def transform_audio_transcription_response(
self,
raw_response: httpx.Response,
) -> TranscriptionResponse:
raise NotImplementedError(
"AudioTranscriptionConfig does not need a response transformation for audio transcription models"
)
def transform_request(
self,
model: str,
@ -84,3 +110,65 @@ class BaseAudioTranscriptionConfig(BaseConfig, ABC):
raise NotImplementedError(
"AudioTranscriptionConfig does not need a response transformation for audio transcription models"
)
def get_provider_specific_params(
self,
model: str,
optional_params: dict,
openai_params: List[OpenAIAudioTranscriptionOptionalParams],
) -> dict:
"""
Get provider specific parameters that are not OpenAI compatible
eg. if user passes `diarize=True`, we need to pass `diarize` to the provider
but `diarize` is not an OpenAI parameter, so we need to handle it here
"""
provider_specific_params = {}
for key, value in optional_params.items():
# Skip None values
if value is None:
continue
# Skip excluded parameters
if self._should_exclude_param(
param_name=key,
model=model,
):
continue
# Add the parameter to the provider specific params
provider_specific_params[key] = value
return provider_specific_params
def _should_exclude_param(
self,
param_name: str,
model: str,
) -> bool:
"""
Determines if a parameter should be excluded from the query string.
Args:
param_name: Parameter name
model: Model name
Returns:
True if the parameter should be excluded
"""
# Parameters that are handled elsewhere or not relevant to Deepgram API
excluded_params = {
"model", # Already in the URL path
"OPENAI_TRANSCRIPTION_PARAMS", # Internal litellm parameter
}
# Skip if it's an excluded parameter
if param_name in excluded_params:
return True
# Skip if it's an OpenAI-specific parameter that we handle separately
if param_name in self.get_supported_openai_params(model):
return True
return False

View file

@ -1004,11 +1004,16 @@ class BaseLLMHTTPHandler:
api_base: Optional[str],
headers: Optional[Dict[str, Any]],
provider_config: BaseAudioTranscriptionConfig,
) -> Tuple[dict, str, Optional[bytes], Optional[dict]]:
) -> Tuple[dict, str, Union[dict, bytes, None], Optional[dict]]:
"""
Shared logic for preparing audio transcription requests.
Returns: (headers, complete_url, binary_data, json_data)
"""
Returns: (headers, complete_url, data, files)
"""
# Handle the response based on type
from litellm.llms.base_llm.audio_transcription.transformation import (
AudioTranscriptionRequestData,
)
headers = provider_config.validate_environment(
api_key=api_key,
headers=headers or {},
@ -1026,32 +1031,33 @@ class BaseLLMHTTPHandler:
litellm_params=litellm_params,
)
# Handle the audio file based on type
data = provider_config.transform_audio_transcription_request(
# Transform the request to get data
transformed_result = provider_config.transform_audio_transcription_request(
model=model,
audio_file=audio_file,
optional_params=optional_params,
litellm_params=litellm_params,
)
binary_data: Optional[bytes] = None
json_data: Optional[dict] = None
if isinstance(data, bytes):
binary_data = data
else:
json_data = data
# All providers now return AudioTranscriptionRequestData
if not isinstance(transformed_result, AudioTranscriptionRequestData):
raise ValueError(f"Provider {provider_config.__class__.__name__} must return AudioTranscriptionRequestData")
data = transformed_result.data
files = transformed_result.files
## LOGGING
logging_obj.pre_call(
input=optional_params.get("query", ""),
api_key=api_key,
additional_args={
"complete_input_dict": {},
"complete_input_dict": data or {},
"api_base": complete_url,
"headers": headers,
},
)
return headers, complete_url, binary_data, json_data
return headers, complete_url, data, files
def _transform_audio_transcription_response(
self,
@ -1064,18 +1070,9 @@ class BaseLLMHTTPHandler:
api_key: Optional[str],
) -> TranscriptionResponse:
"""Shared logic for transforming audio transcription responses."""
if isinstance(provider_config, litellm.DeepgramAudioTranscriptionConfig):
return provider_config.transform_audio_transcription_response(
model=model,
raw_response=response,
model_response=model_response,
logging_obj=logging_obj,
request_data={},
optional_params=optional_params,
litellm_params={},
api_key=api_key,
)
return model_response
return provider_config.transform_audio_transcription_response(
raw_response=response,
)
def audio_transcriptions(
self,
@ -1122,8 +1119,8 @@ class BaseLLMHTTPHandler:
(
headers,
complete_url,
binary_data,
json_data,
data,
files,
) = self._prepare_audio_transcription_request(
model=model,
audio_file=audio_file,
@ -1140,12 +1137,13 @@ class BaseLLMHTTPHandler:
client = _get_httpx_client()
try:
# Make the POST request
# Make the POST request - clean and simple, always use data and files
response = client.post(
url=complete_url,
headers=headers,
content=binary_data,
json=json_data,
data=data,
files=files,
json=data if files is None and isinstance(data, dict) else None, # Use json param only when no files and data is dict
timeout=timeout,
)
except Exception as e:
@ -1187,8 +1185,8 @@ class BaseLLMHTTPHandler:
(
headers,
complete_url,
binary_data,
json_data,
data,
files,
) = self._prepare_audio_transcription_request(
model=model,
audio_file=audio_file,
@ -1210,12 +1208,13 @@ class BaseLLMHTTPHandler:
async_httpx_client = client
try:
# Make the async POST request
# Make the async POST request - clean and simple, always use data and files
response = await async_httpx_client.post(
url=complete_url,
headers=headers,
content=binary_data,
json=json_data,
data=data,
files=files,
json=data if files is None and isinstance(data, dict) else None, # Use json param only when no files and data is dict
timeout=timeout,
)
except Exception as e:

View file

@ -2,12 +2,12 @@
Translates from OpenAI's `/v1/audio/transcriptions` to Deepgram's `/v1/listen`
"""
import io
from typing import List, Optional, Union
from urllib.parse import urlencode
from httpx import Headers, Response
from litellm.litellm_core_utils.audio_utils.utils import process_audio_file
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import (
@ -17,8 +17,8 @@ from litellm.types.llms.openai import (
from litellm.types.utils import FileTypes, TranscriptionResponse
from ...base_llm.audio_transcription.transformation import (
AudioTranscriptionRequestData,
BaseAudioTranscriptionConfig,
LiteLLMLoggingObj,
)
from ..common_utils import DeepgramException
@ -55,59 +55,31 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
audio_file: FileTypes,
optional_params: dict,
litellm_params: dict,
) -> Union[dict, bytes]:
) -> AudioTranscriptionRequestData:
"""
Processes the audio file input based on its type and returns the binary data.
Processes the audio file input based on its type and returns AudioTranscriptionRequestData.
For Deepgram, the binary audio data is sent directly as the request body.
Args:
audio_file: Can be a file path (str), a tuple (filename, file_content), or binary data (bytes).
Returns:
The binary data of the audio file.
AudioTranscriptionRequestData with binary data and no files.
"""
binary_data: bytes # Explicitly declare the type
# Handle the audio file based on type
if isinstance(audio_file, str):
# If it's a file path
with open(audio_file, "rb") as f:
binary_data = f.read() # `f.read()` always returns `bytes`
elif isinstance(audio_file, tuple):
# Handle tuple case
_, file_content = audio_file[:2]
if isinstance(file_content, str):
with open(file_content, "rb") as f:
binary_data = f.read() # `f.read()` always returns `bytes`
elif isinstance(file_content, bytes):
binary_data = file_content
else:
raise TypeError(
f"Unexpected type in tuple: {type(file_content)}. Expected str or bytes."
)
elif isinstance(audio_file, bytes):
# Assume it's already binary data
binary_data = audio_file
elif isinstance(audio_file, io.BufferedReader) or isinstance(
audio_file, io.BytesIO
):
# Handle file-like objects
binary_data = audio_file.read()
else:
raise TypeError(f"Unsupported type for audio_file: {type(audio_file)}")
return binary_data
# Use common utility to process the audio file
processed_audio = process_audio_file(audio_file)
# Return structured data with binary content and no files
# For Deepgram, we send binary data directly as request body
return AudioTranscriptionRequestData(
data=processed_audio.file_content,
files=None
)
def transform_audio_transcription_response(
self,
model: str,
raw_response: Response,
model_response: TranscriptionResponse,
logging_obj: LiteLLMLoggingObj,
request_data: dict,
optional_params: dict,
litellm_params: dict,
api_key: Optional[str] = None,
) -> TranscriptionResponse:
"""
Transforms the raw response from Deepgram to the TranscriptionResponse format
@ -178,36 +150,6 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
return url
def _should_exclude_param(
self,
param_name: str,
model: str,
) -> bool:
"""
Determines if a parameter should be excluded from the query string.
Args:
param_name: Parameter name
model: Model name
Returns:
True if the parameter should be excluded
"""
# Parameters that are handled elsewhere or not relevant to Deepgram API
excluded_params = {
"model", # Already in the URL path
"OPENAI_TRANSCRIPTION_PARAMS", # Internal litellm parameter
}
# Skip if it's an excluded parameter
if param_name in excluded_params:
return True
# Skip if it's an OpenAI-specific parameter that we handle separately
if param_name in self.get_supported_openai_params(model):
return True
return False
def _format_param_value(self, value) -> str:
"""
@ -235,19 +177,13 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
Dictionary of filtered and formatted query parameters
"""
query_params = {}
provider_specific_params = self.get_provider_specific_params(
optional_params=optional_params,
model=model,
openai_params=self.get_supported_openai_params(model)
)
for key, value in optional_params.items():
# Skip None values
if value is None:
continue
# Skip excluded parameters
if self._should_exclude_param(
param_name=key,
model=model,
):
continue
for key, value in provider_specific_params.items():
# Format and add the parameter
formatted_value = self._format_param_value(value)
query_params[key] = formatted_value

View file

@ -0,0 +1,197 @@
"""
Translates from OpenAI's `/v1/audio/transcriptions` to ElevenLabs's `/v1/speech-to-text`
"""
from typing import List, Optional, Union
from httpx import Headers, Response
import litellm
from litellm.litellm_core_utils.audio_utils.utils import process_audio_file
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import (
AllMessageValues,
OpenAIAudioTranscriptionOptionalParams,
)
from litellm.types.utils import FileTypes, TranscriptionResponse
from ...base_llm.audio_transcription.transformation import (
AudioTranscriptionRequestData,
BaseAudioTranscriptionConfig,
)
from ..common_utils import ElevenLabsException
class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
@property
def custom_llm_provider(self) -> str:
return litellm.LlmProviders.ELEVENLABS.value
def get_supported_openai_params(
self, model: str
) -> List[OpenAIAudioTranscriptionOptionalParams]:
return ["language", "temperature"]
def map_openai_params(
self,
non_default_params: dict,
optional_params: dict,
model: str,
drop_params: bool,
) -> dict:
supported_params = self.get_supported_openai_params(model)
for k, v in non_default_params.items():
if k in supported_params:
if k == "language":
# Map OpenAI language format to ElevenLabs language_code
optional_params["language_code"] = v
else:
optional_params[k] = v
return optional_params
def get_error_class(
self, error_message: str, status_code: int, headers: Union[dict, Headers]
) -> BaseLLMException:
return ElevenLabsException(
message=error_message, status_code=status_code, headers=headers
)
def transform_audio_transcription_request(
self,
model: str,
audio_file: FileTypes,
optional_params: dict,
litellm_params: dict,
) -> AudioTranscriptionRequestData:
"""
Transforms the audio transcription request for ElevenLabs API.
Returns AudioTranscriptionRequestData with both form data and files.
Returns:
AudioTranscriptionRequestData: Structured data with form data and files
"""
# Use common utility to process the audio file
processed_audio = process_audio_file(audio_file)
# Prepare form data
form_data = {"model_id": model}
#########################################################
# Add OpenAI Compatible Parameters
#########################################################
for key, value in optional_params.items():
if key in self.get_supported_openai_params(model) and value is not None:
# Convert values to strings for form data, but skip None values
form_data[key] = str(value)
#########################################################
# Add Provider Specific Parameters
#########################################################
provider_specific_params = self.get_provider_specific_params(
model=model,
optional_params=optional_params,
openai_params=self.get_supported_openai_params(model)
)
for key, value in provider_specific_params.items():
form_data[key] = str(value)
#########################################################
#########################################################
# Prepare files
files = {"file": (processed_audio.filename, processed_audio.file_content, processed_audio.content_type)}
return AudioTranscriptionRequestData(
data=form_data,
files=files
)
def transform_audio_transcription_response(
self,
raw_response: Response,
) -> TranscriptionResponse:
"""
Transforms the raw response from ElevenLabs to the TranscriptionResponse format
"""
try:
response_json = raw_response.json()
# Extract the main transcript text
text = response_json.get("text", "")
# Create TranscriptionResponse object
response = TranscriptionResponse(text=text)
# Add additional metadata matching OpenAI format
response["task"] = "transcribe"
response["language"] = response_json.get("language_code", "unknown")
# Map ElevenLabs words to OpenAI format
if "words" in response_json:
response["words"] = []
for word_data in response_json["words"]:
# Only include actual words, skip spacing and audio events
if word_data.get("type") == "word":
response["words"].append({
"word": word_data.get("text", ""),
"start": word_data.get("start", 0),
"end": word_data.get("end", 0)
})
# Store full response in hidden params
response._hidden_params = response_json
return response
except Exception as e:
raise ValueError(
f"Error transforming ElevenLabs response: {str(e)}\nResponse: {raw_response.text}"
)
def get_complete_url(
self,
api_base: Optional[str],
api_key: Optional[str],
model: str,
optional_params: dict,
litellm_params: dict,
stream: Optional[bool] = None,
) -> str:
if api_base is None:
api_base = (
get_secret_str("ELEVENLABS_API_BASE") or "https://api.elevenlabs.io"
)
api_base = api_base.rstrip("/") # Remove trailing slash if present
# ElevenLabs speech-to-text endpoint
url = f"{api_base}/v1/speech-to-text"
return url
def validate_environment(
self,
headers: dict,
model: str,
messages: List[AllMessageValues],
optional_params: dict,
litellm_params: dict,
api_key: Optional[str] = None,
api_base: Optional[str] = None,
) -> dict:
api_key = api_key or get_secret_str("ELEVENLABS_API_KEY")
if api_key is None:
raise ValueError(
"ElevenLabs API key is required. Set ELEVENLABS_API_KEY environment variable."
)
auth_header = {
"xi-api-key": api_key,
}
headers.update(auth_header)
return headers

View file

@ -0,0 +1,5 @@
from litellm.llms.base_llm.chat.transformation import BaseLLMException
class ElevenLabsException(BaseLLMException):
pass

View file

@ -100,7 +100,7 @@ class OpenAIAudioTranscription(OpenAIChatCompletion):
litellm_params=litellm_params,
)
if isinstance(data, bytes):
if not isinstance(data, dict):
raise ValueError("OpenAI transformation route requires a dict")
else:
data = {"model": model, "file": audio_file, **optional_params}

View file

@ -4937,7 +4937,7 @@ def transcription(
provider_config=provider_config,
litellm_params=litellm_params_dict,
)
elif custom_llm_provider == "deepgram":
elif custom_llm_provider in [LlmProviders.DEEPGRAM.value, LlmProviders.ELEVENLABS.value]:
response = base_llm_http_handler.audio_transcriptions(
model=model,
audio_file=file,
@ -4959,7 +4959,7 @@ def transcription(
logging_obj=litellm_logging_obj,
api_base=api_base,
api_key=api_key,
custom_llm_provider="deepgram",
custom_llm_provider=custom_llm_provider,
headers={},
provider_config=provider_config,
)

View file

@ -15704,5 +15704,35 @@
"metadata": {
"notes": "Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"
}
},
"elevenlabs/scribe_v1": {
"mode": "audio_transcription",
"input_cost_per_second": 0.0000611,
"output_cost_per_second": 0.0,
"litellm_provider": "elevenlabs",
"supported_endpoints": [
"/v1/audio/transcriptions"
],
"source": "https://elevenlabs.io/pricing",
"metadata": {
"original_pricing_per_hour": 0.22,
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
"notes": "ElevenLabs Scribe v1 - state-of-the-art speech recognition model with 99 language support"
}
},
"elevenlabs/scribe_v1_experimental": {
"mode": "audio_transcription",
"input_cost_per_second": 0.0000611,
"output_cost_per_second": 0.0,
"litellm_provider": "elevenlabs",
"supported_endpoints": [
"/v1/audio/transcriptions"
],
"source": "https://elevenlabs.io/pricing",
"metadata": {
"original_pricing_per_hour": 0.22,
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
"notes": "ElevenLabs Scribe v1 experimental - enhanced version of the main Scribe model"
}
}
}

Binary file not shown.

After

Width:  |  Height:  |  Size: 35 KiB

View file

@ -2300,6 +2300,7 @@ class LlmProviders(str, Enum):
NEBIUS = "nebius"
INFINITY = "infinity"
DEEPGRAM = "deepgram"
ELEVENLABS = "elevenlabs"
NOVITA = "novita"
AIOHTTP_OPENAI = "aiohttp_openai"
LANGFUSE = "langfuse"

View file

@ -6864,6 +6864,11 @@ class ProviderConfigManager:
return litellm.FireworksAIAudioTranscriptionConfig()
elif litellm.LlmProviders.DEEPGRAM == provider:
return litellm.DeepgramAudioTranscriptionConfig()
elif litellm.LlmProviders.ELEVENLABS == provider:
from litellm.llms.elevenlabs.audio_transcription.transformation import (
ElevenLabsAudioTranscriptionConfig,
)
return ElevenLabsAudioTranscriptionConfig()
elif litellm.LlmProviders.OPENAI == provider:
if "gpt-4o" in model:
return litellm.OpenAIGPTAudioTranscriptionConfig()

View file

@ -15704,5 +15704,35 @@
"metadata": {
"notes": "Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"
}
},
"elevenlabs/scribe_v1": {
"mode": "audio_transcription",
"input_cost_per_second": 0.0000611,
"output_cost_per_second": 0.0,
"litellm_provider": "elevenlabs",
"supported_endpoints": [
"/v1/audio/transcriptions"
],
"source": "https://elevenlabs.io/pricing",
"metadata": {
"original_pricing_per_hour": 0.22,
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
"notes": "ElevenLabs Scribe v1 - state-of-the-art speech recognition model with 99 language support"
}
},
"elevenlabs/scribe_v1_experimental": {
"mode": "audio_transcription",
"input_cost_per_second": 0.0000611,
"output_cost_per_second": 0.0,
"litellm_provider": "elevenlabs",
"supported_endpoints": [
"/v1/audio/transcriptions"
],
"source": "https://elevenlabs.io/pricing",
"metadata": {
"original_pricing_per_hour": 0.22,
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
"notes": "ElevenLabs Scribe v1 experimental - enhanced version of the main Scribe model"
}
}
}

View file

@ -0,0 +1,111 @@
import os
import sys
import pytest
from unittest.mock import patch, MagicMock
import httpx
sys.path.insert(
0, os.path.abspath("../..")
) # Adds the parent directory to the system path
import litellm
from base_audio_transcription_unit_tests import BaseLLMAudioTranscriptionTest
class TestElevenLabsAudioTranscription(BaseLLMAudioTranscriptionTest):
def get_base_audio_transcription_call_args(self) -> dict:
return {
"model": "elevenlabs/scribe_v1",
}
def get_custom_llm_provider(self) -> litellm.LlmProviders:
return litellm.LlmProviders.ELEVENLABS
def test_elevenlabs_diarize_parameter_passthrough(self):
"""
Test that provider-specific parameters like diarize=True get passed through
to the ElevenLabs request form data.
"""
# Mock successful response
mock_response = MagicMock()
mock_response.status_code = 200
mock_response.text = '{"text": "Four score and seven years ago", "language_code": "en"}'
mock_response.json.return_value = {
"text": "Four score and seven years ago",
"language_code": "en",
"words": [
{"type": "word", "text": "Four", "start": 0.0, "end": 0.5},
{"type": "word", "text": "score", "start": 0.5, "end": 1.0}
]
}
# Create a mock audio file
audio_content = b"fake audio data"
captured_request_data = {}
def mock_post(*args, **kwargs):
# Capture the request data for verification
captured_request_data.update({
'url': kwargs.get('url'),
'data': kwargs.get('data'),
'files': kwargs.get('files'),
'headers': kwargs.get('headers'),
'json': kwargs.get('json')
})
return mock_response
# Mock the HTTPHandler.post method which is what actually makes the request
from litellm.llms.custom_httpx.http_handler import HTTPHandler
with patch.object(HTTPHandler, 'post', side_effect=mock_post):
try:
result = litellm.transcription(
model="elevenlabs/scribe_v1",
file=audio_content,
diarize=True, # This should be passed through to the form data
language="en", # This should be mapped to language_code
temperature=0.5, # This should also be passed through
custom_param="test_value" # This should also be passed through
)
# Verify the request was made with correct form data
assert 'speech-to-text' in captured_request_data['url']
# Check that form data contains the expected parameters
form_data = captured_request_data['data']
assert form_data is not None, "Form data should not be None"
print(f"✅ Captured form data: {form_data}")
# Check basic required parameters
assert 'model_id' in form_data, "model_id should be in form data"
assert form_data['model_id'] == 'scribe_v1', f"Expected model_id 'scribe_v1', got {form_data['model_id']}"
# Check that diarize parameter is passed through
assert 'diarize' in form_data, f"diarize should be in form data. Got: {list(form_data.keys())}"
assert form_data['diarize'] == 'True', f"Expected diarize='True', got {form_data['diarize']}"
# Check that OpenAI language parameter is mapped correctly
assert 'language_code' in form_data, "language_code should be in form data"
assert form_data['language_code'] == 'en', f"Expected language_code='en', got {form_data['language_code']}"
# Check that temperature is passed through
assert 'temperature' in form_data, "temperature should be in form data"
assert form_data['temperature'] == '0.5', f"Expected temperature='0.5', got {form_data['temperature']}"
# Check that custom parameters are passed through
assert 'custom_param' in form_data, "custom_param should be in form data"
assert form_data['custom_param'] == 'test_value', f"Expected custom_param='test_value', got {form_data['custom_param']}"
# Check that files are included
files = captured_request_data['files']
assert files is not None, "Files should not be None"
assert 'file' in files, "file should be in files"
print("✅ All parameter passthrough tests passed!")
except Exception as e:
print(f"❌ Test failed: {e}")
print(f"Captured request data: {captured_request_data}")
raise

View file

@ -0,0 +1,208 @@
"""
Test the audio utils functionality in litellm_core_utils/audio_utils/utils.py
"""
import io
import os
import tempfile
from unittest.mock import mock_open, patch
import pytest
from litellm.litellm_core_utils.audio_utils.utils import (
ProcessedAudioFile,
get_audio_file_for_health_check,
get_audio_file_name,
process_audio_file,
)
class TestProcessAudioFile:
"""Test the process_audio_file function with various input types"""
def test_process_bytes_input(self):
"""Test processing raw bytes input"""
audio_data = b"fake audio data"
result = process_audio_file(audio_data)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == audio_data
assert result.filename == "audio.wav"
assert result.content_type == "audio/wav"
def test_process_bytearray_input(self):
"""Test processing bytearray input"""
audio_data = bytearray(b"fake audio data")
result = process_audio_file(audio_data)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == bytes(audio_data)
assert result.filename == "audio.wav"
assert result.content_type == "audio/wav"
def test_process_file_path_input(self):
"""Test processing file path input"""
test_content = b"test audio content"
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as temp_file:
temp_file.write(test_content)
temp_file_path = temp_file.name
try:
result = process_audio_file(temp_file_path)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == test_content
assert result.filename == os.path.basename(temp_file_path)
assert result.content_type == "audio/mpeg" # .mp3 should map to audio/mpeg
finally:
os.unlink(temp_file_path)
def test_process_tuple_input_with_bytes(self):
"""Test processing tuple input with bytes content"""
filename = "test.wav"
audio_data = b"fake audio data"
audio_tuple = (filename, audio_data)
result = process_audio_file(audio_tuple)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == audio_data
assert result.filename == filename
assert result.content_type == "audio/wav"
def test_process_tuple_input_with_file_path(self):
"""Test processing tuple input with file path content"""
test_content = b"test audio content"
with tempfile.NamedTemporaryFile(suffix=".flac", delete=False) as temp_file:
temp_file.write(test_content)
temp_file_path = temp_file.name
try:
filename = "custom_name.flac"
audio_tuple = (filename, temp_file_path)
result = process_audio_file(audio_tuple)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == test_content
assert result.filename == filename
assert result.content_type == "audio/flac"
finally:
os.unlink(temp_file_path)
def test_process_file_like_object(self):
"""Test processing file-like object input"""
test_content = b"test audio content"
file_obj = io.BytesIO(test_content)
file_obj.name = "test_audio.ogg"
result = process_audio_file(file_obj)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == test_content
assert result.filename == "test_audio.ogg"
assert result.content_type == "audio/ogg"
# Verify file pointer was reset
assert file_obj.tell() == 0
def test_process_file_like_object_without_name(self):
"""Test processing file-like object without name attribute"""
test_content = b"test audio content"
file_obj = io.BytesIO(test_content)
result = process_audio_file(file_obj)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == test_content
assert result.filename == "audio.wav"
assert result.content_type == "audio/wav"
def test_process_tuple_with_file_like_object(self):
"""Test processing tuple with file-like object as content"""
test_content = b"test audio content"
file_obj = io.BytesIO(test_content)
filename = "custom.mp3"
audio_tuple = (filename, file_obj)
result = process_audio_file(audio_tuple)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == test_content
assert result.filename == filename
assert result.content_type == "audio/mpeg"
# Verify file pointer was reset
assert file_obj.tell() == 0
def test_mime_type_detection_various_extensions(self):
"""Test MIME type detection for various audio file extensions"""
test_cases = [
("test.wav", "audio/wav"),
("test.mp3", "audio/mpeg"),
("test.flac", "audio/flac"),
("test.ogg", "audio/ogg"),
("test.aac", "audio/aac"),
("test.m4a", "audio/x-m4a"),
]
for filename, expected_mime_type in test_cases:
audio_tuple = (filename, b"fake content")
result = process_audio_file(audio_tuple)
assert result.content_type == expected_mime_type, f"Failed for {filename}"
def test_mime_type_fallback_for_unknown_extension(self):
"""Test MIME type fallback for unknown file extensions"""
audio_tuple = ("test.unknown", b"fake content")
result = process_audio_file(audio_tuple)
assert result.content_type == "audio/wav" # Should fallback to default
def test_process_pathlike_object(self):
"""Test processing os.PathLike object"""
test_content = b"test audio content"
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as temp_file:
temp_file.write(test_content)
temp_file_path = temp_file.name
try:
# Convert to pathlib.Path
from pathlib import Path
path_obj = Path(temp_file_path)
result = process_audio_file(path_obj)
assert isinstance(result, ProcessedAudioFile)
assert result.file_content == test_content
assert result.filename == os.path.basename(temp_file_path)
assert result.content_type == "audio/wav"
finally:
os.unlink(temp_file_path)
def test_invalid_input_type(self):
"""Test that invalid input types raise ValueError"""
with pytest.raises(ValueError, match="Unsupported audio_file type"):
process_audio_file(123) # Invalid type
def test_invalid_tuple_length(self):
"""Test that tuple with less than 2 elements raises ValueError"""
with pytest.raises(ValueError, match="Tuple must have at least 2 elements"):
process_audio_file(("only_one_element",))
def test_invalid_tuple_content_type(self):
"""Test that tuple with unsupported content type raises ValueError"""
with pytest.raises(ValueError, match="Unsupported content type in tuple"):
process_audio_file(("filename", 123)) # Invalid content type
def test_tuple_with_none_filename(self):
"""Test tuple with None filename gets default name"""
audio_tuple = (None, b"fake content")
result = process_audio_file(audio_tuple)
assert result.filename == "audio.wav"
assert result.content_type == "audio/wav"

View file

@ -10,6 +10,9 @@ sys.path.insert(
) # Adds the parent directory to the system path
import litellm
from litellm.llms.base_llm.audio_transcription.transformation import (
AudioTranscriptionRequestData,
)
from litellm.llms.deepgram.audio_transcription.transformation import (
DeepgramAudioTranscriptionConfig,
)
@ -49,12 +52,21 @@ def test_file():
def test_audio_file_handling(fixture_name, request):
handler = DeepgramAudioTranscriptionConfig()
(audio_file, expected_output) = request.getfixturevalue(fixture_name)
assert expected_output == handler.transform_audio_transcription_request(
result = handler.transform_audio_transcription_request(
model="deepseek-audio-transcription",
audio_file=audio_file,
optional_params={},
litellm_params={},
)
# Check that result is AudioTranscriptionRequestData
assert isinstance(result, AudioTranscriptionRequestData)
# Check that data matches expected output
assert result.data == expected_output
# Check that files is None for Deepgram (binary data)
assert result.files is None
def test_get_complete_url_basic():

Binary file not shown.

After

Width:  |  Height:  |  Size: 35 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 35 KiB

View file

@ -266,6 +266,12 @@ const PROVIDER_CREDENTIAL_FIELDS: Record<Providers, ProviderCredentialField[]> =
type: "password",
required: true
}],
[Providers.ElevenLabs]: [{
key: "api_key",
label: "API Key",
type: "password",
required: true
}],
[Providers.Google_AI_Studio]: [{
key: "api_key",
label: "API Key",

View file

@ -27,7 +27,8 @@ export enum Providers {
Openrouter = "Openrouter",
FireworksAI = "Fireworks AI",
Triton = "Triton",
Deepgram = "Deepgram"
Deepgram = "Deepgram",
ElevenLabs = "ElevenLabs"
}
@ -57,7 +58,8 @@ export const provider_map: Record<string, string> = {
Openrouter: "openrouter",
FireworksAI: "fireworks_ai",
Triton: "triton",
Deepgram: "deepgram"
Deepgram: "deepgram",
ElevenLabs: "elevenlabs"
};
const asset_logos_folder = '/ui/assets/logos/';
@ -88,7 +90,8 @@ export const providerLogoMap: Record<string, string> = {
[Providers.Vertex_AI]: `${asset_logos_folder}google.svg`,
[Providers.xAI]: `${asset_logos_folder}xai.svg`,
[Providers.Triton]: `${asset_logos_folder}nvidia_triton.png`,
[Providers.Deepgram]: `${asset_logos_folder}deepgram.png`
[Providers.Deepgram]: `${asset_logos_folder}deepgram.png`,
[Providers.ElevenLabs]: `${asset_logos_folder}elevenlabs.png`
};
export const getProviderLogoAndName = (providerValue: string): { logo: string, displayName: string } => {