From aa52fd59d96521a577f055ea4e81a24be37701e5 Mon Sep 17 00:00:00 2001 From: Tech Carrement Date: Tue, 30 Jun 2026 17:01:24 +0200 Subject: [PATCH] fix(elevenlabs): support multichannel transcription in native /audio/transcriptions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two changes to litellm/llms/elevenlabs/audio_transcription/transformation.py so ElevenLabs Scribe's multichannel speech-to-text works through the native transcription path: 1. Request — serialize boolean form-data values as lowercase `true`/`false` instead of Python's `str(True)` -> `"True"`. ElevenLabs (and its SDK, via httpx) expects lowercase, so flags like `use_multi_channel` / `diarize` were being sent as `"True"` and silently ignored. 2. Response — when `use_multi_channel=true`, ElevenLabs returns a per-channel `transcripts` array (each entry has a `channel_index` and its own `words`) instead of a flat top-level `text`. Surface that `transcripts` array (and `audio_duration_secs`) on the TranscriptionResponse so per-channel/speaker data isn't dropped. These ride through to the client as extra fields because TranscriptionResponse allows extras; single-channel responses are unchanged. Co-Authored-By: Claude Opus 4.8 --- .../audio_transcription/transformation.py | 33 +++++++++++++++++-- 1 file changed, 31 insertions(+), 2 deletions(-) diff --git a/litellm/llms/elevenlabs/audio_transcription/transformation.py b/litellm/llms/elevenlabs/audio_transcription/transformation.py index 68d1b5e16dd..c31e5df1b7f 100644 --- a/litellm/llms/elevenlabs/audio_transcription/transformation.py +++ b/litellm/llms/elevenlabs/audio_transcription/transformation.py @@ -23,6 +23,20 @@ from ...base_llm.audio_transcription.transformation import ( from ..common_utils import ElevenLabsException +def _to_form_value(value: object) -> str: + """Serialize a multipart form-field value the way ElevenLabs expects. + + httpx (which the ElevenLabs SDK uses) encodes booleans as lowercase + ``true``/``false``. Python's ``str(True)`` yields ``"True"``, which the + ElevenLabs API does not recognize — so a boolean flag such as + ``use_multi_channel`` or ``diarize`` would be silently ignored. Normalize + bools to lowercase; everything else is stringified as before. + """ + if isinstance(value, bool): + return "true" if value else "false" + return str(value) + + class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig): @property def custom_llm_provider(self) -> str: @@ -79,7 +93,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig): for key, value in optional_params.items(): if key in self.get_supported_openai_params(model) and value is not None: # Convert values to strings for form data, but skip None values - form_data[key] = str(value) + form_data[key] = _to_form_value(value) ######################################################### # Add Provider Specific Parameters @@ -91,7 +105,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig): ) for key, value in provider_specific_params.items(): - form_data[key] = str(value) + form_data[key] = _to_form_value(value) ######################################################### ######################################################### @@ -140,6 +154,21 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig): } ) + # Surface ElevenLabs multichannel output. With `use_multi_channel=true` + # the response carries a per-channel `transcripts` array (each entry has a + # `channel_index` plus its own `words`) instead of a flat top-level `text`. + # Pass it through verbatim so callers can attribute words to a speaker by + # channel; without this the per-channel data is dropped. + transcripts = response_json.get("transcripts") + if isinstance(transcripts, list): + response["transcripts"] = transcripts + + # Carry the billed audio duration (present on both single- and + # multi-channel responses) so callers can attribute cost. + duration = response_json.get("audio_duration_secs") + if duration is not None: + response["audio_duration_secs"] = duration + # Store full response in hidden params response._hidden_params = response_json