mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(elevenlabs): support multichannel transcription in native /audio/transcriptions
Two changes to litellm/llms/elevenlabs/audio_transcription/transformation.py so ElevenLabs Scribe's multichannel speech-to-text works through the native transcription path: 1. Request — serialize boolean form-data values as lowercase `true`/`false` instead of Python's `str(True)` -> `"True"`. ElevenLabs (and its SDK, via httpx) expects lowercase, so flags like `use_multi_channel` / `diarize` were being sent as `"True"` and silently ignored. 2. Response — when `use_multi_channel=true`, ElevenLabs returns a per-channel `transcripts` array (each entry has a `channel_index` and its own `words`) instead of a flat top-level `text`. Surface that `transcripts` array (and `audio_duration_secs`) on the TranscriptionResponse so per-channel/speaker data isn't dropped. These ride through to the client as extra fields because TranscriptionResponse allows extras; single-channel responses are unchanged. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
parent
0ade44f4da
commit
aa52fd59d9
1 changed files with 31 additions and 2 deletions
|
|
@ -23,6 +23,20 @@ from ...base_llm.audio_transcription.transformation import (
|
|||
from ..common_utils import ElevenLabsException
|
||||
|
||||
|
||||
def _to_form_value(value: object) -> str:
|
||||
"""Serialize a multipart form-field value the way ElevenLabs expects.
|
||||
|
||||
httpx (which the ElevenLabs SDK uses) encodes booleans as lowercase
|
||||
``true``/``false``. Python's ``str(True)`` yields ``"True"``, which the
|
||||
ElevenLabs API does not recognize — so a boolean flag such as
|
||||
``use_multi_channel`` or ``diarize`` would be silently ignored. Normalize
|
||||
bools to lowercase; everything else is stringified as before.
|
||||
"""
|
||||
if isinstance(value, bool):
|
||||
return "true" if value else "false"
|
||||
return str(value)
|
||||
|
||||
|
||||
class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
||||
@property
|
||||
def custom_llm_provider(self) -> str:
|
||||
|
|
@ -79,7 +93,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
for key, value in optional_params.items():
|
||||
if key in self.get_supported_openai_params(model) and value is not None:
|
||||
# Convert values to strings for form data, but skip None values
|
||||
form_data[key] = str(value)
|
||||
form_data[key] = _to_form_value(value)
|
||||
|
||||
#########################################################
|
||||
# Add Provider Specific Parameters
|
||||
|
|
@ -91,7 +105,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
)
|
||||
|
||||
for key, value in provider_specific_params.items():
|
||||
form_data[key] = str(value)
|
||||
form_data[key] = _to_form_value(value)
|
||||
#########################################################
|
||||
#########################################################
|
||||
|
||||
|
|
@ -140,6 +154,21 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
}
|
||||
)
|
||||
|
||||
# Surface ElevenLabs multichannel output. With `use_multi_channel=true`
|
||||
# the response carries a per-channel `transcripts` array (each entry has a
|
||||
# `channel_index` plus its own `words`) instead of a flat top-level `text`.
|
||||
# Pass it through verbatim so callers can attribute words to a speaker by
|
||||
# channel; without this the per-channel data is dropped.
|
||||
transcripts = response_json.get("transcripts")
|
||||
if isinstance(transcripts, list):
|
||||
response["transcripts"] = transcripts
|
||||
|
||||
# Carry the billed audio duration (present on both single- and
|
||||
# multi-channel responses) so callers can attribute cost.
|
||||
duration = response_json.get("audio_duration_secs")
|
||||
if duration is not None:
|
||||
response["audio_duration_secs"] = duration
|
||||
|
||||
# Store full response in hidden params
|
||||
response._hidden_params = response_json
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue