fix(elevenlabs): support multichannel transcription in native /audio/transcriptions

Two changes to litellm/llms/elevenlabs/audio_transcription/transformation.py so
ElevenLabs Scribe's multichannel speech-to-text works through the native
transcription path:

1. Request — serialize boolean form-data values as lowercase `true`/`false`
   instead of Python's `str(True)` -> `"True"`. ElevenLabs (and its SDK, via
   httpx) expects lowercase, so flags like `use_multi_channel` / `diarize` were
   being sent as `"True"` and silently ignored.

2. Response — when `use_multi_channel=true`, ElevenLabs returns a per-channel
   `transcripts` array (each entry has a `channel_index` and its own `words`)
   instead of a flat top-level `text`. Surface that `transcripts` array (and
   `audio_duration_secs`) on the TranscriptionResponse so per-channel/speaker
   data isn't dropped. These ride through to the client as extra fields because
   TranscriptionResponse allows extras; single-channel responses are unchanged.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Tech Carrement 2026-06-30 17:01:24 +02:00
parent 0ade44f4da
commit aa52fd59d9

View file

@ -23,6 +23,20 @@ from ...base_llm.audio_transcription.transformation import (
from ..common_utils import ElevenLabsException
def _to_form_value(value: object) -> str:
"""Serialize a multipart form-field value the way ElevenLabs expects.
httpx (which the ElevenLabs SDK uses) encodes booleans as lowercase
``true``/``false``. Python's ``str(True)`` yields ``"True"``, which the
ElevenLabs API does not recognize — so a boolean flag such as
``use_multi_channel`` or ``diarize`` would be silently ignored. Normalize
bools to lowercase; everything else is stringified as before.
"""
if isinstance(value, bool):
return "true" if value else "false"
return str(value)
class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
@property
def custom_llm_provider(self) -> str:
@ -79,7 +93,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
for key, value in optional_params.items():
if key in self.get_supported_openai_params(model) and value is not None:
# Convert values to strings for form data, but skip None values
form_data[key] = str(value)
form_data[key] = _to_form_value(value)
#########################################################
# Add Provider Specific Parameters
@ -91,7 +105,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
)
for key, value in provider_specific_params.items():
form_data[key] = str(value)
form_data[key] = _to_form_value(value)
#########################################################
#########################################################
@ -140,6 +154,21 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
}
)
# Surface ElevenLabs multichannel output. With `use_multi_channel=true`
# the response carries a per-channel `transcripts` array (each entry has a
# `channel_index` plus its own `words`) instead of a flat top-level `text`.
# Pass it through verbatim so callers can attribute words to a speaker by
# channel; without this the per-channel data is dropped.
transcripts = response_json.get("transcripts")
if isinstance(transcripts, list):
response["transcripts"] = transcripts
# Carry the billed audio duration (present on both single- and
# multi-channel responses) so callers can attribute cost.
duration = response_json.get("audio_duration_secs")
if duration is not None:
response["audio_duration_secs"] = duration
# Store full response in hidden params
response._hidden_params = response_json