mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
[Feat] Add Eleven Labs - Speech To Text Support on LiteLLM (#12119)
* add ELEVENLABS as a provider * add deepgram to main.py * add ElevenLabsException * add ElevenLabsAudioTranscriptionConfig * add transform_audio_transcription_response * TestElevenLabsAudioTranscription * add elevenlabs/scribe_v1 to model cost map * add ElevenLabsAudioTranscriptionConfig * add AudioTranscriptionRequestData * add ElevenLabs transform * use AudioTranscriptionRequestData * refactoring fixes * add ProcessedAudioFile util for reading audio files * test_elevenlabs_diarize_parameter_passthrough * docs eleven labs * docs fixes * fix code qa checks * fixes - audio transcription * ui - add ElevenLabs logo * add elevenlabs logo * docs - ElevenLabs * test fix elevenlabs
This commit is contained in:
parent
041db0268c
commit
ebf6395bc1
24 changed files with 1109 additions and 131 deletions
231
docs/my-website/docs/providers/elevenlabs.md
Normal file
231
docs/my-website/docs/providers/elevenlabs.md
Normal file
|
|
@ -0,0 +1,231 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# ElevenLabs
|
||||
|
||||
ElevenLabs provides high-quality AI voice technology, including speech-to-text capabilities through their transcription API.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | ElevenLabs offers advanced AI voice technology with speech-to-text transcription capabilities that support multiple languages and speaker diarization. |
|
||||
| Provider Route on LiteLLM | `elevenlabs/` |
|
||||
| Provider Doc | [ElevenLabs API ↗](https://elevenlabs.io/docs/api-reference) |
|
||||
| Supported Endpoints | `/audio/transcriptions` |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="basic" label="Basic Usage">
|
||||
|
||||
```python showLineNumbers title="Basic audio transcription with ElevenLabs"
|
||||
import litellm
|
||||
|
||||
# Transcribe audio file
|
||||
with open("audio.mp3", "rb") as audio_file:
|
||||
response = litellm.transcription(
|
||||
model="elevenlabs/scribe_v1",
|
||||
file=audio_file,
|
||||
api_key="your-elevenlabs-api-key" # or set ELEVENLABS_API_KEY env var
|
||||
)
|
||||
|
||||
print(response.text)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="advanced" label="Advanced Features">
|
||||
|
||||
```python showLineNumbers title="Audio transcription with advanced features"
|
||||
import litellm
|
||||
|
||||
# Transcribe with speaker diarization and language specification
|
||||
with open("audio.wav", "rb") as audio_file:
|
||||
response = litellm.transcription(
|
||||
model="elevenlabs/scribe_v1",
|
||||
file=audio_file,
|
||||
language="en", # Language hint (maps to language_code)
|
||||
temperature=0.3, # Control randomness in transcription
|
||||
diarize=True, # Enable speaker diarization
|
||||
api_key="your-elevenlabs-api-key"
|
||||
)
|
||||
|
||||
print(f"Transcription: {response.text}")
|
||||
print(f"Language: {response.language}")
|
||||
|
||||
# Access word-level timestamps if available
|
||||
if hasattr(response, 'words') and response.words:
|
||||
for word_info in response.words:
|
||||
print(f"Word: {word_info['word']}, Start: {word_info['start']}, End: {word_info['end']}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="async" label="Async Usage">
|
||||
|
||||
```python showLineNumbers title="Async audio transcription"
|
||||
import litellm
|
||||
import asyncio
|
||||
|
||||
async def transcribe_audio():
|
||||
with open("audio.mp3", "rb") as audio_file:
|
||||
response = await litellm.atranscription(
|
||||
model="elevenlabs/scribe_v1",
|
||||
file=audio_file,
|
||||
api_key="your-elevenlabs-api-key"
|
||||
)
|
||||
|
||||
return response.text
|
||||
|
||||
# Run async transcription
|
||||
result = asyncio.run(transcribe_audio())
|
||||
print(result)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### LiteLLM Proxy
|
||||
|
||||
#### 1. Configure your proxy
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config-yaml" label="config.yaml">
|
||||
|
||||
```yaml showLineNumbers title="ElevenLabs configuration in config.yaml"
|
||||
model_list:
|
||||
- model_name: elevenlabs-transcription
|
||||
litellm_params:
|
||||
model: elevenlabs/scribe_v1
|
||||
api_key: os.environ/ELEVENLABS_API_KEY
|
||||
|
||||
general_settings:
|
||||
master_key: your-master-key
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="env-vars" label="Environment Variables">
|
||||
|
||||
```bash showLineNumbers title="Required environment variables"
|
||||
export ELEVENLABS_API_KEY="your-elevenlabs-api-key"
|
||||
export LITELLM_MASTER_KEY="your-master-key"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### 2. Start the proxy
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM proxy server"
|
||||
litellm --config config.yaml
|
||||
|
||||
# Proxy will be available at http://localhost:4000
|
||||
```
|
||||
|
||||
#### 3. Make transcription requests
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Audio transcription with curl"
|
||||
curl http://localhost:4000/v1/audio/transcriptions \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-H "Content-Type: multipart/form-data" \
|
||||
-F file="@audio.mp3" \
|
||||
-F model="elevenlabs-transcription" \
|
||||
-F language="en" \
|
||||
-F temperature="0.3"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai-sdk" label="OpenAI Python SDK">
|
||||
|
||||
```python showLineNumbers title="Using OpenAI SDK with LiteLLM proxy"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your LiteLLM proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
# Transcribe audio file
|
||||
with open("audio.mp3", "rb") as audio_file:
|
||||
response = client.audio.transcriptions.create(
|
||||
model="elevenlabs-transcription",
|
||||
file=audio_file,
|
||||
language="en",
|
||||
temperature=0.3,
|
||||
# ElevenLabs-specific parameters
|
||||
diarize=True,
|
||||
speaker_boost=True,
|
||||
custom_vocabulary="technical,AI,machine learning"
|
||||
)
|
||||
|
||||
print(response.text)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="javascript" label="JavaScript/Node.js">
|
||||
|
||||
```javascript showLineNumbers title="Audio transcription with JavaScript"
|
||||
import OpenAI from 'openai';
|
||||
import fs from 'fs';
|
||||
|
||||
const openai = new OpenAI({
|
||||
baseURL: 'http://localhost:4000',
|
||||
apiKey: 'your-litellm-api-key'
|
||||
});
|
||||
|
||||
async function transcribeAudio() {
|
||||
const response = await openai.audio.transcriptions.create({
|
||||
file: fs.createReadStream('audio.mp3'),
|
||||
model: 'elevenlabs-transcription',
|
||||
language: 'en',
|
||||
temperature: 0.3,
|
||||
diarize: true,
|
||||
speaker_boost: true
|
||||
});
|
||||
|
||||
console.log(response.text);
|
||||
}
|
||||
|
||||
transcribeAudio();
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Response Format
|
||||
|
||||
ElevenLabs returns transcription responses in OpenAI-compatible format:
|
||||
|
||||
```json showLineNumbers title="Example transcription response"
|
||||
{
|
||||
"text": "Hello, this is a sample transcription with multiple speakers.",
|
||||
"task": "transcribe",
|
||||
"language": "en",
|
||||
"words": [
|
||||
{
|
||||
"word": "Hello",
|
||||
"start": 0.0,
|
||||
"end": 0.5
|
||||
},
|
||||
{
|
||||
"word": "this",
|
||||
"start": 0.5,
|
||||
"end": 0.8
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Common Issues
|
||||
|
||||
1. **Invalid API Key**: Ensure `ELEVENLABS_API_KEY` is set correctly
|
||||
|
||||
|
||||
|
|
@ -415,6 +415,7 @@ const sidebars = {
|
|||
"providers/groq",
|
||||
"providers/github",
|
||||
"providers/deepseek",
|
||||
"providers/elevenlabs",
|
||||
"providers/fireworks_ai",
|
||||
"providers/clarifai",
|
||||
"providers/vllm",
|
||||
|
|
|
|||
|
|
@ -478,6 +478,7 @@ nscale_models: List = []
|
|||
nebius_models: List = []
|
||||
nebius_embedding_models: List = []
|
||||
deepgram_models: List = []
|
||||
elevenlabs_models: List = []
|
||||
|
||||
|
||||
def is_bedrock_pricing_only_model(key: str) -> bool:
|
||||
|
|
@ -651,6 +652,8 @@ def add_known_models():
|
|||
featherless_ai_models.append(key)
|
||||
elif value.get("litellm_provider") == "deepgram":
|
||||
deepgram_models.append(key)
|
||||
elif value.get("litellm_provider") == "elevenlabs":
|
||||
elevenlabs_models.append(key)
|
||||
|
||||
|
||||
add_known_models()
|
||||
|
|
@ -733,6 +736,7 @@ model_list = (
|
|||
+ featherless_ai_models
|
||||
+ nscale_models
|
||||
+ deepgram_models
|
||||
+ elevenlabs_models
|
||||
)
|
||||
|
||||
model_list_set = set(model_list)
|
||||
|
|
@ -797,6 +801,7 @@ models_by_provider: dict = {
|
|||
"nscale": nscale_models,
|
||||
"featherless_ai": featherless_ai_models,
|
||||
"deepgram": deepgram_models,
|
||||
"elevenlabs": elevenlabs_models,
|
||||
}
|
||||
|
||||
# mapping for those models which have larger equivalents
|
||||
|
|
|
|||
|
|
@ -3,10 +3,110 @@ Utils used for litellm.transcription() and litellm.atranscription()
|
|||
"""
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
|
||||
from litellm.types.files import get_file_mime_type_from_extension
|
||||
from litellm.types.utils import FileTypes
|
||||
|
||||
|
||||
@dataclass
|
||||
class ProcessedAudioFile:
|
||||
"""
|
||||
Processed audio file data.
|
||||
|
||||
Attributes:
|
||||
file_content: The binary content of the audio file
|
||||
filename: The filename (extracted or generated)
|
||||
content_type: The MIME type of the audio file
|
||||
"""
|
||||
file_content: bytes
|
||||
filename: str
|
||||
content_type: str
|
||||
|
||||
|
||||
def process_audio_file(audio_file: FileTypes) -> ProcessedAudioFile:
|
||||
"""
|
||||
Common utility function to process audio files for audio transcription APIs.
|
||||
|
||||
Handles various input types:
|
||||
- File paths (str, os.PathLike)
|
||||
- Raw bytes/bytearray
|
||||
- Tuples (filename, content, optional content_type)
|
||||
- File-like objects with read() method
|
||||
|
||||
Args:
|
||||
audio_file: The audio file input in various formats
|
||||
|
||||
Returns:
|
||||
ProcessedAudioFile: Structured data with file content, filename, and content type
|
||||
|
||||
Raises:
|
||||
ValueError: If audio_file type is unsupported or content cannot be extracted
|
||||
"""
|
||||
file_content = None
|
||||
filename = None
|
||||
|
||||
if isinstance(audio_file, (bytes, bytearray)):
|
||||
# Raw bytes
|
||||
filename = 'audio.wav'
|
||||
file_content = bytes(audio_file)
|
||||
elif isinstance(audio_file, (str, os.PathLike)):
|
||||
# File path or PathLike
|
||||
file_path = str(audio_file)
|
||||
with open(file_path, 'rb') as f:
|
||||
file_content = f.read()
|
||||
filename = file_path.split('/')[-1]
|
||||
elif isinstance(audio_file, tuple):
|
||||
# Tuple format: (filename, content, content_type) or (filename, content)
|
||||
if len(audio_file) >= 2:
|
||||
filename = audio_file[0] or 'audio.wav'
|
||||
content = audio_file[1]
|
||||
if isinstance(content, (bytes, bytearray)):
|
||||
file_content = bytes(content)
|
||||
elif isinstance(content, (str, os.PathLike)):
|
||||
# File path or PathLike
|
||||
with open(str(content), 'rb') as f:
|
||||
file_content = f.read()
|
||||
elif hasattr(content, 'read'):
|
||||
# File-like object
|
||||
file_content = content.read()
|
||||
if hasattr(content, 'seek'):
|
||||
content.seek(0)
|
||||
else:
|
||||
raise ValueError(f"Unsupported content type in tuple: {type(content)}")
|
||||
else:
|
||||
raise ValueError("Tuple must have at least 2 elements: (filename, content)")
|
||||
elif hasattr(audio_file, 'read') and not isinstance(audio_file, (str, bytes, bytearray, tuple, os.PathLike)):
|
||||
# File-like object (IO) - check this after all other types
|
||||
filename = getattr(audio_file, 'name', 'audio.wav')
|
||||
file_content = audio_file.read() # type: ignore
|
||||
# Reset file pointer if possible
|
||||
if hasattr(audio_file, 'seek'):
|
||||
audio_file.seek(0) # type: ignore
|
||||
else:
|
||||
raise ValueError(f"Unsupported audio_file type: {type(audio_file)}")
|
||||
|
||||
if file_content is None:
|
||||
raise ValueError("Could not extract file content from audio_file")
|
||||
|
||||
# Determine content type using LiteLLM's file type utilities
|
||||
content_type = 'audio/wav' # Default fallback
|
||||
if filename:
|
||||
try:
|
||||
# Extract extension from filename
|
||||
extension = filename.split('.')[-1].lower() if '.' in filename else 'wav'
|
||||
content_type = get_file_mime_type_from_extension(extension)
|
||||
except ValueError:
|
||||
# If extension is not recognized, fallback to audio/wav
|
||||
content_type = 'audio/wav'
|
||||
|
||||
return ProcessedAudioFile(
|
||||
file_content=file_content,
|
||||
filename=filename,
|
||||
content_type=content_type
|
||||
)
|
||||
|
||||
|
||||
def get_audio_file_name(file_obj: FileTypes) -> str:
|
||||
"""
|
||||
Safely get the name of a file-like object or return its string representation.
|
||||
|
|
|
|||
|
|
@ -252,6 +252,16 @@ def get_supported_openai_params( # noqa: PLR0915
|
|||
model=model
|
||||
)
|
||||
)
|
||||
elif custom_llm_provider == "elevenlabs":
|
||||
if request_type == "transcription":
|
||||
from litellm.llms.elevenlabs.audio_transcription.transformation import (
|
||||
ElevenLabsAudioTranscriptionConfig,
|
||||
)
|
||||
return (
|
||||
ElevenLabsAudioTranscriptionConfig().get_supported_openai_params(
|
||||
model=model
|
||||
)
|
||||
)
|
||||
elif custom_llm_provider in litellm._custom_providers:
|
||||
if request_type == "chat_completion":
|
||||
provider_config = litellm.ProviderConfigManager.get_provider_chat_config(
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
from abc import ABC, abstractmethod
|
||||
from typing import TYPE_CHECKING, Any, List, Optional, Union
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union
|
||||
|
||||
import httpx
|
||||
|
||||
|
|
@ -8,7 +9,7 @@ from litellm.types.llms.openai import (
|
|||
AllMessageValues,
|
||||
OpenAIAudioTranscriptionOptionalParams,
|
||||
)
|
||||
from litellm.types.utils import FileTypes, ModelResponse
|
||||
from litellm.types.utils import FileTypes, ModelResponse, TranscriptionResponse
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
|
||||
|
|
@ -18,6 +19,21 @@ else:
|
|||
LiteLLMLoggingObj = Any
|
||||
|
||||
|
||||
@dataclass
|
||||
class AudioTranscriptionRequestData:
|
||||
"""
|
||||
Structured data for audio transcription requests.
|
||||
|
||||
Attributes:
|
||||
data: The request data (form data for multipart, json data for regular requests)
|
||||
files: Optional files dict for multipart form data
|
||||
content_type: Optional content type override
|
||||
"""
|
||||
data: Union[dict, bytes]
|
||||
files: Optional[dict] = None
|
||||
content_type: Optional[str] = None
|
||||
|
||||
|
||||
class BaseAudioTranscriptionConfig(BaseConfig, ABC):
|
||||
@abstractmethod
|
||||
def get_supported_openai_params(
|
||||
|
|
@ -50,11 +66,21 @@ class BaseAudioTranscriptionConfig(BaseConfig, ABC):
|
|||
audio_file: FileTypes,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
) -> Union[dict, bytes]:
|
||||
) -> Union[AudioTranscriptionRequestData, Dict]:
|
||||
raise NotImplementedError(
|
||||
"AudioTranscriptionConfig needs a request transformation for audio transcription models"
|
||||
)
|
||||
|
||||
|
||||
|
||||
def transform_audio_transcription_response(
|
||||
self,
|
||||
raw_response: httpx.Response,
|
||||
) -> TranscriptionResponse:
|
||||
raise NotImplementedError(
|
||||
"AudioTranscriptionConfig does not need a response transformation for audio transcription models"
|
||||
)
|
||||
|
||||
def transform_request(
|
||||
self,
|
||||
model: str,
|
||||
|
|
@ -84,3 +110,65 @@ class BaseAudioTranscriptionConfig(BaseConfig, ABC):
|
|||
raise NotImplementedError(
|
||||
"AudioTranscriptionConfig does not need a response transformation for audio transcription models"
|
||||
)
|
||||
|
||||
|
||||
def get_provider_specific_params(
|
||||
self,
|
||||
model: str,
|
||||
optional_params: dict,
|
||||
openai_params: List[OpenAIAudioTranscriptionOptionalParams],
|
||||
) -> dict:
|
||||
"""
|
||||
Get provider specific parameters that are not OpenAI compatible
|
||||
|
||||
eg. if user passes `diarize=True`, we need to pass `diarize` to the provider
|
||||
but `diarize` is not an OpenAI parameter, so we need to handle it here
|
||||
"""
|
||||
provider_specific_params = {}
|
||||
for key, value in optional_params.items():
|
||||
# Skip None values
|
||||
if value is None:
|
||||
continue
|
||||
|
||||
# Skip excluded parameters
|
||||
if self._should_exclude_param(
|
||||
param_name=key,
|
||||
model=model,
|
||||
):
|
||||
continue
|
||||
|
||||
# Add the parameter to the provider specific params
|
||||
provider_specific_params[key] = value
|
||||
|
||||
return provider_specific_params
|
||||
|
||||
def _should_exclude_param(
|
||||
self,
|
||||
param_name: str,
|
||||
model: str,
|
||||
) -> bool:
|
||||
"""
|
||||
Determines if a parameter should be excluded from the query string.
|
||||
|
||||
Args:
|
||||
param_name: Parameter name
|
||||
model: Model name
|
||||
|
||||
Returns:
|
||||
True if the parameter should be excluded
|
||||
"""
|
||||
# Parameters that are handled elsewhere or not relevant to Deepgram API
|
||||
excluded_params = {
|
||||
"model", # Already in the URL path
|
||||
"OPENAI_TRANSCRIPTION_PARAMS", # Internal litellm parameter
|
||||
}
|
||||
|
||||
# Skip if it's an excluded parameter
|
||||
if param_name in excluded_params:
|
||||
return True
|
||||
|
||||
# Skip if it's an OpenAI-specific parameter that we handle separately
|
||||
if param_name in self.get_supported_openai_params(model):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
|
|
|||
|
|
@ -1004,11 +1004,16 @@ class BaseLLMHTTPHandler:
|
|||
api_base: Optional[str],
|
||||
headers: Optional[Dict[str, Any]],
|
||||
provider_config: BaseAudioTranscriptionConfig,
|
||||
) -> Tuple[dict, str, Optional[bytes], Optional[dict]]:
|
||||
) -> Tuple[dict, str, Union[dict, bytes, None], Optional[dict]]:
|
||||
"""
|
||||
Shared logic for preparing audio transcription requests.
|
||||
Returns: (headers, complete_url, binary_data, json_data)
|
||||
"""
|
||||
Returns: (headers, complete_url, data, files)
|
||||
"""
|
||||
# Handle the response based on type
|
||||
from litellm.llms.base_llm.audio_transcription.transformation import (
|
||||
AudioTranscriptionRequestData,
|
||||
)
|
||||
|
||||
headers = provider_config.validate_environment(
|
||||
api_key=api_key,
|
||||
headers=headers or {},
|
||||
|
|
@ -1026,32 +1031,33 @@ class BaseLLMHTTPHandler:
|
|||
litellm_params=litellm_params,
|
||||
)
|
||||
|
||||
# Handle the audio file based on type
|
||||
data = provider_config.transform_audio_transcription_request(
|
||||
# Transform the request to get data
|
||||
transformed_result = provider_config.transform_audio_transcription_request(
|
||||
model=model,
|
||||
audio_file=audio_file,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
)
|
||||
binary_data: Optional[bytes] = None
|
||||
json_data: Optional[dict] = None
|
||||
if isinstance(data, bytes):
|
||||
binary_data = data
|
||||
else:
|
||||
json_data = data
|
||||
|
||||
# All providers now return AudioTranscriptionRequestData
|
||||
if not isinstance(transformed_result, AudioTranscriptionRequestData):
|
||||
raise ValueError(f"Provider {provider_config.__class__.__name__} must return AudioTranscriptionRequestData")
|
||||
|
||||
data = transformed_result.data
|
||||
files = transformed_result.files
|
||||
|
||||
## LOGGING
|
||||
logging_obj.pre_call(
|
||||
input=optional_params.get("query", ""),
|
||||
api_key=api_key,
|
||||
additional_args={
|
||||
"complete_input_dict": {},
|
||||
"complete_input_dict": data or {},
|
||||
"api_base": complete_url,
|
||||
"headers": headers,
|
||||
},
|
||||
)
|
||||
|
||||
return headers, complete_url, binary_data, json_data
|
||||
return headers, complete_url, data, files
|
||||
|
||||
def _transform_audio_transcription_response(
|
||||
self,
|
||||
|
|
@ -1064,18 +1070,9 @@ class BaseLLMHTTPHandler:
|
|||
api_key: Optional[str],
|
||||
) -> TranscriptionResponse:
|
||||
"""Shared logic for transforming audio transcription responses."""
|
||||
if isinstance(provider_config, litellm.DeepgramAudioTranscriptionConfig):
|
||||
return provider_config.transform_audio_transcription_response(
|
||||
model=model,
|
||||
raw_response=response,
|
||||
model_response=model_response,
|
||||
logging_obj=logging_obj,
|
||||
request_data={},
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
api_key=api_key,
|
||||
)
|
||||
return model_response
|
||||
return provider_config.transform_audio_transcription_response(
|
||||
raw_response=response,
|
||||
)
|
||||
|
||||
def audio_transcriptions(
|
||||
self,
|
||||
|
|
@ -1122,8 +1119,8 @@ class BaseLLMHTTPHandler:
|
|||
(
|
||||
headers,
|
||||
complete_url,
|
||||
binary_data,
|
||||
json_data,
|
||||
data,
|
||||
files,
|
||||
) = self._prepare_audio_transcription_request(
|
||||
model=model,
|
||||
audio_file=audio_file,
|
||||
|
|
@ -1140,12 +1137,13 @@ class BaseLLMHTTPHandler:
|
|||
client = _get_httpx_client()
|
||||
|
||||
try:
|
||||
# Make the POST request
|
||||
# Make the POST request - clean and simple, always use data and files
|
||||
response = client.post(
|
||||
url=complete_url,
|
||||
headers=headers,
|
||||
content=binary_data,
|
||||
json=json_data,
|
||||
data=data,
|
||||
files=files,
|
||||
json=data if files is None and isinstance(data, dict) else None, # Use json param only when no files and data is dict
|
||||
timeout=timeout,
|
||||
)
|
||||
except Exception as e:
|
||||
|
|
@ -1187,8 +1185,8 @@ class BaseLLMHTTPHandler:
|
|||
(
|
||||
headers,
|
||||
complete_url,
|
||||
binary_data,
|
||||
json_data,
|
||||
data,
|
||||
files,
|
||||
) = self._prepare_audio_transcription_request(
|
||||
model=model,
|
||||
audio_file=audio_file,
|
||||
|
|
@ -1210,12 +1208,13 @@ class BaseLLMHTTPHandler:
|
|||
async_httpx_client = client
|
||||
|
||||
try:
|
||||
# Make the async POST request
|
||||
# Make the async POST request - clean and simple, always use data and files
|
||||
response = await async_httpx_client.post(
|
||||
url=complete_url,
|
||||
headers=headers,
|
||||
content=binary_data,
|
||||
json=json_data,
|
||||
data=data,
|
||||
files=files,
|
||||
json=data if files is None and isinstance(data, dict) else None, # Use json param only when no files and data is dict
|
||||
timeout=timeout,
|
||||
)
|
||||
except Exception as e:
|
||||
|
|
|
|||
|
|
@ -2,12 +2,12 @@
|
|||
Translates from OpenAI's `/v1/audio/transcriptions` to Deepgram's `/v1/listen`
|
||||
"""
|
||||
|
||||
import io
|
||||
from typing import List, Optional, Union
|
||||
from urllib.parse import urlencode
|
||||
|
||||
from httpx import Headers, Response
|
||||
|
||||
from litellm.litellm_core_utils.audio_utils.utils import process_audio_file
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.llms.openai import (
|
||||
|
|
@ -17,8 +17,8 @@ from litellm.types.llms.openai import (
|
|||
from litellm.types.utils import FileTypes, TranscriptionResponse
|
||||
|
||||
from ...base_llm.audio_transcription.transformation import (
|
||||
AudioTranscriptionRequestData,
|
||||
BaseAudioTranscriptionConfig,
|
||||
LiteLLMLoggingObj,
|
||||
)
|
||||
from ..common_utils import DeepgramException
|
||||
|
||||
|
|
@ -55,59 +55,31 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
audio_file: FileTypes,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
) -> Union[dict, bytes]:
|
||||
) -> AudioTranscriptionRequestData:
|
||||
"""
|
||||
Processes the audio file input based on its type and returns the binary data.
|
||||
Processes the audio file input based on its type and returns AudioTranscriptionRequestData.
|
||||
|
||||
For Deepgram, the binary audio data is sent directly as the request body.
|
||||
|
||||
Args:
|
||||
audio_file: Can be a file path (str), a tuple (filename, file_content), or binary data (bytes).
|
||||
|
||||
Returns:
|
||||
The binary data of the audio file.
|
||||
AudioTranscriptionRequestData with binary data and no files.
|
||||
"""
|
||||
binary_data: bytes # Explicitly declare the type
|
||||
|
||||
# Handle the audio file based on type
|
||||
if isinstance(audio_file, str):
|
||||
# If it's a file path
|
||||
with open(audio_file, "rb") as f:
|
||||
binary_data = f.read() # `f.read()` always returns `bytes`
|
||||
elif isinstance(audio_file, tuple):
|
||||
# Handle tuple case
|
||||
_, file_content = audio_file[:2]
|
||||
if isinstance(file_content, str):
|
||||
with open(file_content, "rb") as f:
|
||||
binary_data = f.read() # `f.read()` always returns `bytes`
|
||||
elif isinstance(file_content, bytes):
|
||||
binary_data = file_content
|
||||
else:
|
||||
raise TypeError(
|
||||
f"Unexpected type in tuple: {type(file_content)}. Expected str or bytes."
|
||||
)
|
||||
elif isinstance(audio_file, bytes):
|
||||
# Assume it's already binary data
|
||||
binary_data = audio_file
|
||||
elif isinstance(audio_file, io.BufferedReader) or isinstance(
|
||||
audio_file, io.BytesIO
|
||||
):
|
||||
# Handle file-like objects
|
||||
binary_data = audio_file.read()
|
||||
|
||||
else:
|
||||
raise TypeError(f"Unsupported type for audio_file: {type(audio_file)}")
|
||||
|
||||
return binary_data
|
||||
# Use common utility to process the audio file
|
||||
processed_audio = process_audio_file(audio_file)
|
||||
|
||||
# Return structured data with binary content and no files
|
||||
# For Deepgram, we send binary data directly as request body
|
||||
return AudioTranscriptionRequestData(
|
||||
data=processed_audio.file_content,
|
||||
files=None
|
||||
)
|
||||
|
||||
def transform_audio_transcription_response(
|
||||
self,
|
||||
model: str,
|
||||
raw_response: Response,
|
||||
model_response: TranscriptionResponse,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
request_data: dict,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
api_key: Optional[str] = None,
|
||||
) -> TranscriptionResponse:
|
||||
"""
|
||||
Transforms the raw response from Deepgram to the TranscriptionResponse format
|
||||
|
|
@ -178,36 +150,6 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
|
||||
return url
|
||||
|
||||
def _should_exclude_param(
|
||||
self,
|
||||
param_name: str,
|
||||
model: str,
|
||||
) -> bool:
|
||||
"""
|
||||
Determines if a parameter should be excluded from the query string.
|
||||
|
||||
Args:
|
||||
param_name: Parameter name
|
||||
model: Model name
|
||||
|
||||
Returns:
|
||||
True if the parameter should be excluded
|
||||
"""
|
||||
# Parameters that are handled elsewhere or not relevant to Deepgram API
|
||||
excluded_params = {
|
||||
"model", # Already in the URL path
|
||||
"OPENAI_TRANSCRIPTION_PARAMS", # Internal litellm parameter
|
||||
}
|
||||
|
||||
# Skip if it's an excluded parameter
|
||||
if param_name in excluded_params:
|
||||
return True
|
||||
|
||||
# Skip if it's an OpenAI-specific parameter that we handle separately
|
||||
if param_name in self.get_supported_openai_params(model):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def _format_param_value(self, value) -> str:
|
||||
"""
|
||||
|
|
@ -235,19 +177,13 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
Dictionary of filtered and formatted query parameters
|
||||
"""
|
||||
query_params = {}
|
||||
provider_specific_params = self.get_provider_specific_params(
|
||||
optional_params=optional_params,
|
||||
model=model,
|
||||
openai_params=self.get_supported_openai_params(model)
|
||||
)
|
||||
|
||||
for key, value in optional_params.items():
|
||||
# Skip None values
|
||||
if value is None:
|
||||
continue
|
||||
|
||||
# Skip excluded parameters
|
||||
if self._should_exclude_param(
|
||||
param_name=key,
|
||||
model=model,
|
||||
):
|
||||
continue
|
||||
|
||||
for key, value in provider_specific_params.items():
|
||||
# Format and add the parameter
|
||||
formatted_value = self._format_param_value(value)
|
||||
query_params[key] = formatted_value
|
||||
|
|
|
|||
197
litellm/llms/elevenlabs/audio_transcription/transformation.py
Normal file
197
litellm/llms/elevenlabs/audio_transcription/transformation.py
Normal file
|
|
@ -0,0 +1,197 @@
|
|||
"""
|
||||
Translates from OpenAI's `/v1/audio/transcriptions` to ElevenLabs's `/v1/speech-to-text`
|
||||
"""
|
||||
|
||||
from typing import List, Optional, Union
|
||||
|
||||
from httpx import Headers, Response
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.audio_utils.utils import process_audio_file
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
OpenAIAudioTranscriptionOptionalParams,
|
||||
)
|
||||
from litellm.types.utils import FileTypes, TranscriptionResponse
|
||||
|
||||
from ...base_llm.audio_transcription.transformation import (
|
||||
AudioTranscriptionRequestData,
|
||||
BaseAudioTranscriptionConfig,
|
||||
)
|
||||
from ..common_utils import ElevenLabsException
|
||||
|
||||
|
||||
class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
||||
@property
|
||||
def custom_llm_provider(self) -> str:
|
||||
return litellm.LlmProviders.ELEVENLABS.value
|
||||
|
||||
def get_supported_openai_params(
|
||||
self, model: str
|
||||
) -> List[OpenAIAudioTranscriptionOptionalParams]:
|
||||
return ["language", "temperature"]
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict,
|
||||
optional_params: dict,
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict:
|
||||
supported_params = self.get_supported_openai_params(model)
|
||||
for k, v in non_default_params.items():
|
||||
if k in supported_params:
|
||||
if k == "language":
|
||||
# Map OpenAI language format to ElevenLabs language_code
|
||||
optional_params["language_code"] = v
|
||||
else:
|
||||
optional_params[k] = v
|
||||
return optional_params
|
||||
|
||||
def get_error_class(
|
||||
self, error_message: str, status_code: int, headers: Union[dict, Headers]
|
||||
) -> BaseLLMException:
|
||||
return ElevenLabsException(
|
||||
message=error_message, status_code=status_code, headers=headers
|
||||
)
|
||||
|
||||
def transform_audio_transcription_request(
|
||||
self,
|
||||
model: str,
|
||||
audio_file: FileTypes,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
) -> AudioTranscriptionRequestData:
|
||||
"""
|
||||
Transforms the audio transcription request for ElevenLabs API.
|
||||
|
||||
Returns AudioTranscriptionRequestData with both form data and files.
|
||||
|
||||
Returns:
|
||||
AudioTranscriptionRequestData: Structured data with form data and files
|
||||
"""
|
||||
|
||||
# Use common utility to process the audio file
|
||||
processed_audio = process_audio_file(audio_file)
|
||||
|
||||
# Prepare form data
|
||||
form_data = {"model_id": model}
|
||||
|
||||
|
||||
#########################################################
|
||||
# Add OpenAI Compatible Parameters
|
||||
#########################################################
|
||||
for key, value in optional_params.items():
|
||||
if key in self.get_supported_openai_params(model) and value is not None:
|
||||
# Convert values to strings for form data, but skip None values
|
||||
form_data[key] = str(value)
|
||||
|
||||
#########################################################
|
||||
# Add Provider Specific Parameters
|
||||
#########################################################
|
||||
provider_specific_params = self.get_provider_specific_params(
|
||||
model=model,
|
||||
optional_params=optional_params,
|
||||
openai_params=self.get_supported_openai_params(model)
|
||||
)
|
||||
|
||||
for key, value in provider_specific_params.items():
|
||||
form_data[key] = str(value)
|
||||
#########################################################
|
||||
#########################################################
|
||||
|
||||
# Prepare files
|
||||
files = {"file": (processed_audio.filename, processed_audio.file_content, processed_audio.content_type)}
|
||||
|
||||
return AudioTranscriptionRequestData(
|
||||
data=form_data,
|
||||
files=files
|
||||
)
|
||||
|
||||
|
||||
def transform_audio_transcription_response(
|
||||
self,
|
||||
raw_response: Response,
|
||||
) -> TranscriptionResponse:
|
||||
"""
|
||||
Transforms the raw response from ElevenLabs to the TranscriptionResponse format
|
||||
"""
|
||||
try:
|
||||
response_json = raw_response.json()
|
||||
|
||||
# Extract the main transcript text
|
||||
text = response_json.get("text", "")
|
||||
|
||||
# Create TranscriptionResponse object
|
||||
response = TranscriptionResponse(text=text)
|
||||
|
||||
# Add additional metadata matching OpenAI format
|
||||
response["task"] = "transcribe"
|
||||
response["language"] = response_json.get("language_code", "unknown")
|
||||
|
||||
# Map ElevenLabs words to OpenAI format
|
||||
if "words" in response_json:
|
||||
response["words"] = []
|
||||
for word_data in response_json["words"]:
|
||||
# Only include actual words, skip spacing and audio events
|
||||
if word_data.get("type") == "word":
|
||||
response["words"].append({
|
||||
"word": word_data.get("text", ""),
|
||||
"start": word_data.get("start", 0),
|
||||
"end": word_data.get("end", 0)
|
||||
})
|
||||
|
||||
# Store full response in hidden params
|
||||
response._hidden_params = response_json
|
||||
|
||||
return response
|
||||
|
||||
except Exception as e:
|
||||
raise ValueError(
|
||||
f"Error transforming ElevenLabs response: {str(e)}\nResponse: {raw_response.text}"
|
||||
)
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
api_base: Optional[str],
|
||||
api_key: Optional[str],
|
||||
model: str,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
stream: Optional[bool] = None,
|
||||
) -> str:
|
||||
if api_base is None:
|
||||
api_base = (
|
||||
get_secret_str("ELEVENLABS_API_BASE") or "https://api.elevenlabs.io"
|
||||
)
|
||||
api_base = api_base.rstrip("/") # Remove trailing slash if present
|
||||
|
||||
# ElevenLabs speech-to-text endpoint
|
||||
url = f"{api_base}/v1/speech-to-text"
|
||||
|
||||
return url
|
||||
|
||||
def validate_environment(
|
||||
self,
|
||||
headers: dict,
|
||||
model: str,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
api_key: Optional[str] = None,
|
||||
api_base: Optional[str] = None,
|
||||
) -> dict:
|
||||
api_key = api_key or get_secret_str("ELEVENLABS_API_KEY")
|
||||
if api_key is None:
|
||||
raise ValueError(
|
||||
"ElevenLabs API key is required. Set ELEVENLABS_API_KEY environment variable."
|
||||
)
|
||||
|
||||
auth_header = {
|
||||
"xi-api-key": api_key,
|
||||
}
|
||||
|
||||
headers.update(auth_header)
|
||||
return headers
|
||||
5
litellm/llms/elevenlabs/common_utils.py
Normal file
5
litellm/llms/elevenlabs/common_utils.py
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
|
||||
class ElevenLabsException(BaseLLMException):
|
||||
pass
|
||||
|
|
@ -100,7 +100,7 @@ class OpenAIAudioTranscription(OpenAIChatCompletion):
|
|||
litellm_params=litellm_params,
|
||||
)
|
||||
|
||||
if isinstance(data, bytes):
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("OpenAI transformation route requires a dict")
|
||||
else:
|
||||
data = {"model": model, "file": audio_file, **optional_params}
|
||||
|
|
|
|||
|
|
@ -4937,7 +4937,7 @@ def transcription(
|
|||
provider_config=provider_config,
|
||||
litellm_params=litellm_params_dict,
|
||||
)
|
||||
elif custom_llm_provider == "deepgram":
|
||||
elif custom_llm_provider in [LlmProviders.DEEPGRAM.value, LlmProviders.ELEVENLABS.value]:
|
||||
response = base_llm_http_handler.audio_transcriptions(
|
||||
model=model,
|
||||
audio_file=file,
|
||||
|
|
@ -4959,7 +4959,7 @@ def transcription(
|
|||
logging_obj=litellm_logging_obj,
|
||||
api_base=api_base,
|
||||
api_key=api_key,
|
||||
custom_llm_provider="deepgram",
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
headers={},
|
||||
provider_config=provider_config,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -15704,5 +15704,35 @@
|
|||
"metadata": {
|
||||
"notes": "Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"
|
||||
}
|
||||
},
|
||||
"elevenlabs/scribe_v1": {
|
||||
"mode": "audio_transcription",
|
||||
"input_cost_per_second": 0.0000611,
|
||||
"output_cost_per_second": 0.0,
|
||||
"litellm_provider": "elevenlabs",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions"
|
||||
],
|
||||
"source": "https://elevenlabs.io/pricing",
|
||||
"metadata": {
|
||||
"original_pricing_per_hour": 0.22,
|
||||
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
|
||||
"notes": "ElevenLabs Scribe v1 - state-of-the-art speech recognition model with 99 language support"
|
||||
}
|
||||
},
|
||||
"elevenlabs/scribe_v1_experimental": {
|
||||
"mode": "audio_transcription",
|
||||
"input_cost_per_second": 0.0000611,
|
||||
"output_cost_per_second": 0.0,
|
||||
"litellm_provider": "elevenlabs",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions"
|
||||
],
|
||||
"source": "https://elevenlabs.io/pricing",
|
||||
"metadata": {
|
||||
"original_pricing_per_hour": 0.22,
|
||||
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
|
||||
"notes": "ElevenLabs Scribe v1 experimental - enhanced version of the main Scribe model"
|
||||
}
|
||||
}
|
||||
}
|
||||
BIN
litellm/proxy/_experimental/out/assets/logos/elevenlabs.png
Normal file
BIN
litellm/proxy/_experimental/out/assets/logos/elevenlabs.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 35 KiB |
|
|
@ -2300,6 +2300,7 @@ class LlmProviders(str, Enum):
|
|||
NEBIUS = "nebius"
|
||||
INFINITY = "infinity"
|
||||
DEEPGRAM = "deepgram"
|
||||
ELEVENLABS = "elevenlabs"
|
||||
NOVITA = "novita"
|
||||
AIOHTTP_OPENAI = "aiohttp_openai"
|
||||
LANGFUSE = "langfuse"
|
||||
|
|
|
|||
|
|
@ -6864,6 +6864,11 @@ class ProviderConfigManager:
|
|||
return litellm.FireworksAIAudioTranscriptionConfig()
|
||||
elif litellm.LlmProviders.DEEPGRAM == provider:
|
||||
return litellm.DeepgramAudioTranscriptionConfig()
|
||||
elif litellm.LlmProviders.ELEVENLABS == provider:
|
||||
from litellm.llms.elevenlabs.audio_transcription.transformation import (
|
||||
ElevenLabsAudioTranscriptionConfig,
|
||||
)
|
||||
return ElevenLabsAudioTranscriptionConfig()
|
||||
elif litellm.LlmProviders.OPENAI == provider:
|
||||
if "gpt-4o" in model:
|
||||
return litellm.OpenAIGPTAudioTranscriptionConfig()
|
||||
|
|
|
|||
|
|
@ -15704,5 +15704,35 @@
|
|||
"metadata": {
|
||||
"notes": "Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"
|
||||
}
|
||||
},
|
||||
"elevenlabs/scribe_v1": {
|
||||
"mode": "audio_transcription",
|
||||
"input_cost_per_second": 0.0000611,
|
||||
"output_cost_per_second": 0.0,
|
||||
"litellm_provider": "elevenlabs",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions"
|
||||
],
|
||||
"source": "https://elevenlabs.io/pricing",
|
||||
"metadata": {
|
||||
"original_pricing_per_hour": 0.22,
|
||||
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
|
||||
"notes": "ElevenLabs Scribe v1 - state-of-the-art speech recognition model with 99 language support"
|
||||
}
|
||||
},
|
||||
"elevenlabs/scribe_v1_experimental": {
|
||||
"mode": "audio_transcription",
|
||||
"input_cost_per_second": 0.0000611,
|
||||
"output_cost_per_second": 0.0,
|
||||
"litellm_provider": "elevenlabs",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions"
|
||||
],
|
||||
"source": "https://elevenlabs.io/pricing",
|
||||
"metadata": {
|
||||
"original_pricing_per_hour": 0.22,
|
||||
"calculation": "$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)",
|
||||
"notes": "ElevenLabs Scribe v1 experimental - enhanced version of the main Scribe model"
|
||||
}
|
||||
}
|
||||
}
|
||||
111
tests/llm_translation/test_elevenlabs.py
Normal file
111
tests/llm_translation/test_elevenlabs.py
Normal file
|
|
@ -0,0 +1,111 @@
|
|||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
from unittest.mock import patch, MagicMock
|
||||
import httpx
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath("../..")
|
||||
) # Adds the parent directory to the system path
|
||||
import litellm
|
||||
from base_audio_transcription_unit_tests import BaseLLMAudioTranscriptionTest
|
||||
|
||||
|
||||
class TestElevenLabsAudioTranscription(BaseLLMAudioTranscriptionTest):
|
||||
def get_base_audio_transcription_call_args(self) -> dict:
|
||||
return {
|
||||
"model": "elevenlabs/scribe_v1",
|
||||
}
|
||||
|
||||
def get_custom_llm_provider(self) -> litellm.LlmProviders:
|
||||
return litellm.LlmProviders.ELEVENLABS
|
||||
|
||||
def test_elevenlabs_diarize_parameter_passthrough(self):
|
||||
"""
|
||||
Test that provider-specific parameters like diarize=True get passed through
|
||||
to the ElevenLabs request form data.
|
||||
"""
|
||||
# Mock successful response
|
||||
mock_response = MagicMock()
|
||||
mock_response.status_code = 200
|
||||
mock_response.text = '{"text": "Four score and seven years ago", "language_code": "en"}'
|
||||
mock_response.json.return_value = {
|
||||
"text": "Four score and seven years ago",
|
||||
"language_code": "en",
|
||||
"words": [
|
||||
{"type": "word", "text": "Four", "start": 0.0, "end": 0.5},
|
||||
{"type": "word", "text": "score", "start": 0.5, "end": 1.0}
|
||||
]
|
||||
}
|
||||
|
||||
# Create a mock audio file
|
||||
audio_content = b"fake audio data"
|
||||
|
||||
captured_request_data = {}
|
||||
|
||||
def mock_post(*args, **kwargs):
|
||||
# Capture the request data for verification
|
||||
captured_request_data.update({
|
||||
'url': kwargs.get('url'),
|
||||
'data': kwargs.get('data'),
|
||||
'files': kwargs.get('files'),
|
||||
'headers': kwargs.get('headers'),
|
||||
'json': kwargs.get('json')
|
||||
})
|
||||
return mock_response
|
||||
|
||||
# Mock the HTTPHandler.post method which is what actually makes the request
|
||||
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
||||
|
||||
with patch.object(HTTPHandler, 'post', side_effect=mock_post):
|
||||
try:
|
||||
result = litellm.transcription(
|
||||
model="elevenlabs/scribe_v1",
|
||||
file=audio_content,
|
||||
diarize=True, # This should be passed through to the form data
|
||||
language="en", # This should be mapped to language_code
|
||||
temperature=0.5, # This should also be passed through
|
||||
custom_param="test_value" # This should also be passed through
|
||||
)
|
||||
|
||||
# Verify the request was made with correct form data
|
||||
assert 'speech-to-text' in captured_request_data['url']
|
||||
|
||||
# Check that form data contains the expected parameters
|
||||
form_data = captured_request_data['data']
|
||||
assert form_data is not None, "Form data should not be None"
|
||||
|
||||
print(f"✅ Captured form data: {form_data}")
|
||||
|
||||
# Check basic required parameters
|
||||
assert 'model_id' in form_data, "model_id should be in form data"
|
||||
assert form_data['model_id'] == 'scribe_v1', f"Expected model_id 'scribe_v1', got {form_data['model_id']}"
|
||||
|
||||
# Check that diarize parameter is passed through
|
||||
assert 'diarize' in form_data, f"diarize should be in form data. Got: {list(form_data.keys())}"
|
||||
assert form_data['diarize'] == 'True', f"Expected diarize='True', got {form_data['diarize']}"
|
||||
|
||||
# Check that OpenAI language parameter is mapped correctly
|
||||
assert 'language_code' in form_data, "language_code should be in form data"
|
||||
assert form_data['language_code'] == 'en', f"Expected language_code='en', got {form_data['language_code']}"
|
||||
|
||||
# Check that temperature is passed through
|
||||
assert 'temperature' in form_data, "temperature should be in form data"
|
||||
assert form_data['temperature'] == '0.5', f"Expected temperature='0.5', got {form_data['temperature']}"
|
||||
|
||||
# Check that custom parameters are passed through
|
||||
assert 'custom_param' in form_data, "custom_param should be in form data"
|
||||
assert form_data['custom_param'] == 'test_value', f"Expected custom_param='test_value', got {form_data['custom_param']}"
|
||||
|
||||
# Check that files are included
|
||||
files = captured_request_data['files']
|
||||
assert files is not None, "Files should not be None"
|
||||
assert 'file' in files, "file should be in files"
|
||||
|
||||
print("✅ All parameter passthrough tests passed!")
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Test failed: {e}")
|
||||
print(f"Captured request data: {captured_request_data}")
|
||||
raise
|
||||
208
tests/test_litellm/litellm_core_utils/test_audio_utils.py
Normal file
208
tests/test_litellm/litellm_core_utils/test_audio_utils.py
Normal file
|
|
@ -0,0 +1,208 @@
|
|||
"""
|
||||
Test the audio utils functionality in litellm_core_utils/audio_utils/utils.py
|
||||
"""
|
||||
|
||||
import io
|
||||
import os
|
||||
import tempfile
|
||||
from unittest.mock import mock_open, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.litellm_core_utils.audio_utils.utils import (
|
||||
ProcessedAudioFile,
|
||||
get_audio_file_for_health_check,
|
||||
get_audio_file_name,
|
||||
process_audio_file,
|
||||
)
|
||||
|
||||
|
||||
class TestProcessAudioFile:
|
||||
"""Test the process_audio_file function with various input types"""
|
||||
|
||||
def test_process_bytes_input(self):
|
||||
"""Test processing raw bytes input"""
|
||||
audio_data = b"fake audio data"
|
||||
result = process_audio_file(audio_data)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == audio_data
|
||||
assert result.filename == "audio.wav"
|
||||
assert result.content_type == "audio/wav"
|
||||
|
||||
def test_process_bytearray_input(self):
|
||||
"""Test processing bytearray input"""
|
||||
audio_data = bytearray(b"fake audio data")
|
||||
result = process_audio_file(audio_data)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == bytes(audio_data)
|
||||
assert result.filename == "audio.wav"
|
||||
assert result.content_type == "audio/wav"
|
||||
|
||||
def test_process_file_path_input(self):
|
||||
"""Test processing file path input"""
|
||||
test_content = b"test audio content"
|
||||
|
||||
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as temp_file:
|
||||
temp_file.write(test_content)
|
||||
temp_file_path = temp_file.name
|
||||
|
||||
try:
|
||||
result = process_audio_file(temp_file_path)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == test_content
|
||||
assert result.filename == os.path.basename(temp_file_path)
|
||||
assert result.content_type == "audio/mpeg" # .mp3 should map to audio/mpeg
|
||||
finally:
|
||||
os.unlink(temp_file_path)
|
||||
|
||||
def test_process_tuple_input_with_bytes(self):
|
||||
"""Test processing tuple input with bytes content"""
|
||||
filename = "test.wav"
|
||||
audio_data = b"fake audio data"
|
||||
audio_tuple = (filename, audio_data)
|
||||
|
||||
result = process_audio_file(audio_tuple)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == audio_data
|
||||
assert result.filename == filename
|
||||
assert result.content_type == "audio/wav"
|
||||
|
||||
def test_process_tuple_input_with_file_path(self):
|
||||
"""Test processing tuple input with file path content"""
|
||||
test_content = b"test audio content"
|
||||
|
||||
with tempfile.NamedTemporaryFile(suffix=".flac", delete=False) as temp_file:
|
||||
temp_file.write(test_content)
|
||||
temp_file_path = temp_file.name
|
||||
|
||||
try:
|
||||
filename = "custom_name.flac"
|
||||
audio_tuple = (filename, temp_file_path)
|
||||
|
||||
result = process_audio_file(audio_tuple)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == test_content
|
||||
assert result.filename == filename
|
||||
assert result.content_type == "audio/flac"
|
||||
finally:
|
||||
os.unlink(temp_file_path)
|
||||
|
||||
def test_process_file_like_object(self):
|
||||
"""Test processing file-like object input"""
|
||||
test_content = b"test audio content"
|
||||
file_obj = io.BytesIO(test_content)
|
||||
file_obj.name = "test_audio.ogg"
|
||||
|
||||
result = process_audio_file(file_obj)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == test_content
|
||||
assert result.filename == "test_audio.ogg"
|
||||
assert result.content_type == "audio/ogg"
|
||||
|
||||
# Verify file pointer was reset
|
||||
assert file_obj.tell() == 0
|
||||
|
||||
def test_process_file_like_object_without_name(self):
|
||||
"""Test processing file-like object without name attribute"""
|
||||
test_content = b"test audio content"
|
||||
file_obj = io.BytesIO(test_content)
|
||||
|
||||
result = process_audio_file(file_obj)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == test_content
|
||||
assert result.filename == "audio.wav"
|
||||
assert result.content_type == "audio/wav"
|
||||
|
||||
def test_process_tuple_with_file_like_object(self):
|
||||
"""Test processing tuple with file-like object as content"""
|
||||
test_content = b"test audio content"
|
||||
file_obj = io.BytesIO(test_content)
|
||||
|
||||
filename = "custom.mp3"
|
||||
audio_tuple = (filename, file_obj)
|
||||
|
||||
result = process_audio_file(audio_tuple)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == test_content
|
||||
assert result.filename == filename
|
||||
assert result.content_type == "audio/mpeg"
|
||||
|
||||
# Verify file pointer was reset
|
||||
assert file_obj.tell() == 0
|
||||
|
||||
def test_mime_type_detection_various_extensions(self):
|
||||
"""Test MIME type detection for various audio file extensions"""
|
||||
test_cases = [
|
||||
("test.wav", "audio/wav"),
|
||||
("test.mp3", "audio/mpeg"),
|
||||
("test.flac", "audio/flac"),
|
||||
("test.ogg", "audio/ogg"),
|
||||
("test.aac", "audio/aac"),
|
||||
("test.m4a", "audio/x-m4a"),
|
||||
]
|
||||
|
||||
for filename, expected_mime_type in test_cases:
|
||||
audio_tuple = (filename, b"fake content")
|
||||
result = process_audio_file(audio_tuple)
|
||||
assert result.content_type == expected_mime_type, f"Failed for {filename}"
|
||||
|
||||
def test_mime_type_fallback_for_unknown_extension(self):
|
||||
"""Test MIME type fallback for unknown file extensions"""
|
||||
audio_tuple = ("test.unknown", b"fake content")
|
||||
result = process_audio_file(audio_tuple)
|
||||
|
||||
assert result.content_type == "audio/wav" # Should fallback to default
|
||||
|
||||
def test_process_pathlike_object(self):
|
||||
"""Test processing os.PathLike object"""
|
||||
test_content = b"test audio content"
|
||||
|
||||
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as temp_file:
|
||||
temp_file.write(test_content)
|
||||
temp_file_path = temp_file.name
|
||||
|
||||
try:
|
||||
# Convert to pathlib.Path
|
||||
from pathlib import Path
|
||||
path_obj = Path(temp_file_path)
|
||||
|
||||
result = process_audio_file(path_obj)
|
||||
|
||||
assert isinstance(result, ProcessedAudioFile)
|
||||
assert result.file_content == test_content
|
||||
assert result.filename == os.path.basename(temp_file_path)
|
||||
assert result.content_type == "audio/wav"
|
||||
finally:
|
||||
os.unlink(temp_file_path)
|
||||
|
||||
def test_invalid_input_type(self):
|
||||
"""Test that invalid input types raise ValueError"""
|
||||
with pytest.raises(ValueError, match="Unsupported audio_file type"):
|
||||
process_audio_file(123) # Invalid type
|
||||
|
||||
def test_invalid_tuple_length(self):
|
||||
"""Test that tuple with less than 2 elements raises ValueError"""
|
||||
with pytest.raises(ValueError, match="Tuple must have at least 2 elements"):
|
||||
process_audio_file(("only_one_element",))
|
||||
|
||||
def test_invalid_tuple_content_type(self):
|
||||
"""Test that tuple with unsupported content type raises ValueError"""
|
||||
with pytest.raises(ValueError, match="Unsupported content type in tuple"):
|
||||
process_audio_file(("filename", 123)) # Invalid content type
|
||||
|
||||
def test_tuple_with_none_filename(self):
|
||||
"""Test tuple with None filename gets default name"""
|
||||
audio_tuple = (None, b"fake content")
|
||||
result = process_audio_file(audio_tuple)
|
||||
|
||||
assert result.filename == "audio.wav"
|
||||
assert result.content_type == "audio/wav"
|
||||
|
||||
|
|
@ -10,6 +10,9 @@ sys.path.insert(
|
|||
) # Adds the parent directory to the system path
|
||||
|
||||
import litellm
|
||||
from litellm.llms.base_llm.audio_transcription.transformation import (
|
||||
AudioTranscriptionRequestData,
|
||||
)
|
||||
from litellm.llms.deepgram.audio_transcription.transformation import (
|
||||
DeepgramAudioTranscriptionConfig,
|
||||
)
|
||||
|
|
@ -49,12 +52,21 @@ def test_file():
|
|||
def test_audio_file_handling(fixture_name, request):
|
||||
handler = DeepgramAudioTranscriptionConfig()
|
||||
(audio_file, expected_output) = request.getfixturevalue(fixture_name)
|
||||
assert expected_output == handler.transform_audio_transcription_request(
|
||||
result = handler.transform_audio_transcription_request(
|
||||
model="deepseek-audio-transcription",
|
||||
audio_file=audio_file,
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
)
|
||||
|
||||
# Check that result is AudioTranscriptionRequestData
|
||||
assert isinstance(result, AudioTranscriptionRequestData)
|
||||
|
||||
# Check that data matches expected output
|
||||
assert result.data == expected_output
|
||||
|
||||
# Check that files is None for Deepgram (binary data)
|
||||
assert result.files is None
|
||||
|
||||
|
||||
def test_get_complete_url_basic():
|
||||
|
|
|
|||
BIN
ui/litellm-dashboard/out/assets/logos/elevenlabs.png
Normal file
BIN
ui/litellm-dashboard/out/assets/logos/elevenlabs.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 35 KiB |
BIN
ui/litellm-dashboard/public/assets/logos/elevenlabs.png
Normal file
BIN
ui/litellm-dashboard/public/assets/logos/elevenlabs.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 35 KiB |
|
|
@ -266,6 +266,12 @@ const PROVIDER_CREDENTIAL_FIELDS: Record<Providers, ProviderCredentialField[]> =
|
|||
type: "password",
|
||||
required: true
|
||||
}],
|
||||
[Providers.ElevenLabs]: [{
|
||||
key: "api_key",
|
||||
label: "API Key",
|
||||
type: "password",
|
||||
required: true
|
||||
}],
|
||||
[Providers.Google_AI_Studio]: [{
|
||||
key: "api_key",
|
||||
label: "API Key",
|
||||
|
|
|
|||
|
|
@ -27,7 +27,8 @@ export enum Providers {
|
|||
Openrouter = "Openrouter",
|
||||
FireworksAI = "Fireworks AI",
|
||||
Triton = "Triton",
|
||||
Deepgram = "Deepgram"
|
||||
Deepgram = "Deepgram",
|
||||
ElevenLabs = "ElevenLabs"
|
||||
|
||||
}
|
||||
|
||||
|
|
@ -57,7 +58,8 @@ export const provider_map: Record<string, string> = {
|
|||
Openrouter: "openrouter",
|
||||
FireworksAI: "fireworks_ai",
|
||||
Triton: "triton",
|
||||
Deepgram: "deepgram"
|
||||
Deepgram: "deepgram",
|
||||
ElevenLabs: "elevenlabs"
|
||||
};
|
||||
|
||||
const asset_logos_folder = '/ui/assets/logos/';
|
||||
|
|
@ -88,7 +90,8 @@ export const providerLogoMap: Record<string, string> = {
|
|||
[Providers.Vertex_AI]: `${asset_logos_folder}google.svg`,
|
||||
[Providers.xAI]: `${asset_logos_folder}xai.svg`,
|
||||
[Providers.Triton]: `${asset_logos_folder}nvidia_triton.png`,
|
||||
[Providers.Deepgram]: `${asset_logos_folder}deepgram.png`
|
||||
[Providers.Deepgram]: `${asset_logos_folder}deepgram.png`,
|
||||
[Providers.ElevenLabs]: `${asset_logos_folder}elevenlabs.png`
|
||||
};
|
||||
|
||||
export const getProviderLogoAndName = (providerValue: string): { logo: string, displayName: string } => {
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue