Add support of audio transcription for OVHcloud (#17305)

This commit is contained in:
Elias 2025-12-01 21:26:39 -05:00 • committed by GitHub
parent be920d75d3
commit 37ecb03d4f
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 248 additions and 2 deletions

View file

@ -13,7 +13,7 @@ import TabItem from '@theme/TabItem';
| Fallbacks | ✅ | Works between supported models |
| Loadbalancing | ✅ | Works between supported models |
| Guardrails | ✅ | Applies to output transcribed text (non-streaming only) |
| Supported Providers | `openai`, `azure`, `vertex_ai`, `gemini`, `deepgram`, `groq`, `fireworks_ai` | |
| Supported Providers | `openai`, `azure`, `vertex_ai`, `gemini`, `deepgram`, `groq`, `fireworks_ai`, `ovhcloud` | |
## Quick Start
@ -126,6 +126,7 @@ transcript = client.audio.transcriptions.create(
- [Fireworks AI](./providers/fireworks_ai.md#audio-transcription)
- [Groq](./providers/groq.md#speech-to-text---whisper)
- [Deepgram](./providers/deepgram.md)
- [OVHcloud AI Endpoints](./providers/ovhcloud.md)
---

View file

@ -311,6 +311,21 @@ response = embedding(
print(response.data)
```
### Audio Transcription
```python
from litellm import transcription
audio_file = open("path/to/your/audio.wav", "rb")
response = transcription(
model="ovhcloud/whisper-large-v3-turbo",
file=audio_file
)
print(response.text)
```
## Usage with LiteLLM Proxy Server
Here's how to call a OVHCloud AI Endpoints model with the LiteLLM Proxy Server

View file

@ -266,6 +266,15 @@ def get_supported_openai_params( # noqa: PLR0915
model=model
)
)
elif custom_llm_provider == "ovhcloud":
if request_type == "transcription":
from litellm.llms.ovhcloud.audio_transcription.transformation import (
OVHCloudAudioTranscriptionConfig,
)
return OVHCloudAudioTranscriptionConfig().get_supported_openai_params(
model=model
)
elif custom_llm_provider == "elevenlabs":
if request_type == "transcription":
from litellm.llms.elevenlabs.audio_transcription.transformation import (

View file

@ -0,0 +1,156 @@
"""
Support for OVHCloud AI Endpoints `/v1/audio/transcriptions` endpoint.
Our unified API follows the OpenAI standard.
More information on our website: https://endpoints.ai.cloud.ovh.net
"""
from typing import List, Optional, Union
import httpx
from litellm.litellm_core_utils.audio_utils.utils import process_audio_file
from litellm.llms.base_llm.audio_transcription.transformation import (
AudioTranscriptionRequestData,
BaseAudioTranscriptionConfig,
)
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import (
AllMessageValues,
OpenAIAudioTranscriptionOptionalParams,
)
from litellm.types.utils import FileTypes, TranscriptionResponse
from ..utils import OVHCloudException
class OVHCloudAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
def get_supported_openai_params(
self, model: str
) -> List[OpenAIAudioTranscriptionOptionalParams]:
# OVHCloud implements the OpenAI-compatible Whisper interface.
# We pass through the same optional params as the OpenAI Whisper API.
return ["language", "prompt", "response_format", "timestamp_granularities", "temperature"]
def map_openai_params(
self,
non_default_params: dict,
optional_params: dict,
model: str,
drop_params: bool,
) -> dict:
supported_params = self.get_supported_openai_params(model)
for k, v in non_default_params.items():
if k in supported_params:
optional_params[k] = v
return optional_params
def get_complete_url(
self,
api_base: Optional[str],
api_key: Optional[str],
model: str,
optional_params: dict,
litellm_params: dict,
stream: Optional[bool] = None,
) -> str:
api_base = (
"https://oai.endpoints.kepler.ai.cloud.ovh.net/v1"
if api_base is None
else api_base.rstrip("/")
)
complete_url = f"{api_base}/audio/transcriptions"
return complete_url
def get_error_class(
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
) -> BaseLLMException:
return OVHCloudException(
message=error_message,
status_code=status_code,
headers=headers,
)
def validate_environment(
self,
headers: dict,
model: str,
messages: List[AllMessageValues],
optional_params: dict,
litellm_params: dict,
api_key: Optional[str] = None,
api_base: Optional[str] = None,
) -> dict:
if api_key is None:
api_key = get_secret_str("OVHCLOUD_API_KEY")
default_headers = {
"Authorization": f"Bearer {api_key}",
"accept": "application/json",
}
# Caller can override / extend headers if needed
default_headers.update(headers or {})
return default_headers
def transform_audio_transcription_request(
self,
model: str,
audio_file: FileTypes,
optional_params: dict,
litellm_params: dict,
) -> AudioTranscriptionRequestData:
"""
Transform the audio transcription request into OpenAI-compatible form-data.
OVHCloud follows OpenAI's `/audio/transcriptions` format, so we:
- Build a multipart form-data body with `file`, `model`, and optional params
- Let the shared HTTP handler set the proper content-type boundary
"""
processed_audio = process_audio_file(audio_file)
# Base form fields: model + OpenAI-compatible optional params
form_fields: dict = {
"model": model,
}
# Include OpenAI-compatible optional params
for key in self.get_supported_openai_params(model):
value = optional_params.get(key)
if value is not None:
form_fields[key] = value
files = {
"file": (
processed_audio.filename,
processed_audio.file_content,
processed_audio.content_type,
)
}
return AudioTranscriptionRequestData(data=form_fields, files=files)
def transform_audio_transcription_response(
self,
raw_response: httpx.Response,
) -> TranscriptionResponse:
"""
Transform OVHCloud audio transcription response to OpenAI-compatible TranscriptionResponse.
"""
try:
response_json = raw_response.json()
except Exception:
raise OVHCloudException(
message=raw_response.text,
status_code=raw_response.status_code,
headers=raw_response.headers,
)
text = response_json.get("text") or response_json.get("transcript") or ""
response = TranscriptionResponse(text=text)
response._hidden_params = response_json
return response

View file

@ -7384,6 +7384,12 @@ class ProviderConfigManager:
)
return IBMWatsonXAudioTranscriptionConfig()
elif litellm.LlmProviders.OVHCLOUD == provider:
from litellm.llms.ovhcloud.audio_transcription.transformation import (
OVHCloudAudioTranscriptionConfig,
)
return OVHCloudAudioTranscriptionConfig()
return None
@staticmethod

View file

@ -1272,7 +1272,7 @@
"responses": true,
"embeddings": false,
"image_generations": false,
"audio_transcriptions": false,
"audio_transcriptions": true,
"audio_speech": false,
"moderations": false,
"batches": false,

View file

@ -0,0 +1,59 @@
import os
from typing import Dict
import litellm
import pytest
from litellm.llms.base_llm.audio_transcription.transformation import (
BaseAudioTranscriptionConfig,
)
from litellm.utils import ProviderConfigManager
from tests.llm_translation.base_audio_transcription_unit_tests import (
BaseLLMAudioTranscriptionTest,
)
@pytest.mark.skipif(
not os.getenv("OVHCLOUD_API_KEY"),
reason="OVHCLOUD_API_KEY not set, skipping OVHCloud audio transcription tests",
)
class TestOVHCloudAudioTranscription(BaseLLMAudioTranscriptionTest):
def get_base_audio_transcription_call_args(self) -> Dict:
return {
"model": "ovhcloud/whisper-large-v3-turbo",
}
def get_custom_llm_provider(self) -> litellm.LlmProviders:
return litellm.LlmProviders.OVHCLOUD
# Override the async base test with a sync no-op to avoid
# 'async def functions are not natively supported' failures when
# running this file in isolation without pytest-asyncio.
def test_audio_transcription_async(self): # type: ignore[override]
pytest.skip(
"Async audio transcription test for OVHCloud is skipped in this suite; "
"async test plugins (e.g. pytest-asyncio/anyio) are not configured here."
)
@pytest.mark.skipif(
not os.getenv("OVHCLOUD_API_KEY"),
reason="OVHCLOUD_API_KEY not set, skipping OVHCloud audio transcription config test",
)
def test_ovhcloud_audio_transcription_config_installed():
"""
Ensure OVHCloud audio transcription config is registered with ProviderConfigManager.
"""
model = "ovhcloud/whisper-large-v3-turbo"
provider = litellm.LlmProviders.OVHCLOUD
config = ProviderConfigManager.get_provider_audio_transcription_config(
model=model,
provider=provider,
)
assert config is not None
assert isinstance(config, BaseAudioTranscriptionConfig)