From 4b73d62f4e7da3d982571a34438754e21848847c Mon Sep 17 00:00:00 2001 From: Tech Carrement Date: Mon, 6 Jul 2026 22:06:31 +0200 Subject: [PATCH] fix(elevenlabs/stt): synthesize single-channel transcripts for mono responses A mono Scribe response has raw words but no transcripts array; the OpenAI-format words we map are lossy (text->word, spacing/punctuation dropped), so callers that segment bubbles from raw per-word tokens got an empty transcript for mono audio. Wrap the raw words as a one-channel transcripts entry. Non-breaking: multichannel and OpenAI-compat words unchanged. --- .../audio_transcription/transformation.py | 10 ++++++ ...labs_audio_transcription_transformation.py | 33 +++++++++++++++++++ 2 files changed, 43 insertions(+) diff --git a/litellm/llms/elevenlabs/audio_transcription/transformation.py b/litellm/llms/elevenlabs/audio_transcription/transformation.py index c31e5df1b7f..9cd418c5320 100644 --- a/litellm/llms/elevenlabs/audio_transcription/transformation.py +++ b/litellm/llms/elevenlabs/audio_transcription/transformation.py @@ -162,6 +162,16 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig): transcripts = response_json.get("transcripts") if isinstance(transcripts, list): response["transcripts"] = transcripts + elif isinstance(response_json.get("words"), list): + # Single-channel (mono) response: no `transcripts` array, only flat + # top-level `words`. The OpenAI-format `response["words"]` above is + # lossy — it renames `text`->`word` and drops spacing/punctuation and + # audio events. Wrap the RAW words as one channel so callers get the + # same verbatim per-word tokens as multichannel and can segment + # bubbles uniformly; without this a mono transcript loses punctuation. + response["transcripts"] = [ + {"channel_index": 0, "words": response_json["words"]} + ] # Carry the billed audio duration (present on both single- and # multi-channel responses) so callers can attribute cost. diff --git a/tests/test_litellm/llms/elevenlabs/audio_transcription/test_elevenlabs_audio_transcription_transformation.py b/tests/test_litellm/llms/elevenlabs/audio_transcription/test_elevenlabs_audio_transcription_transformation.py index 5c8c8536e4d..cc94a8aee77 100644 --- a/tests/test_litellm/llms/elevenlabs/audio_transcription/test_elevenlabs_audio_transcription_transformation.py +++ b/tests/test_litellm/llms/elevenlabs/audio_transcription/test_elevenlabs_audio_transcription_transformation.py @@ -75,3 +75,36 @@ def test_response_single_channel_is_unchanged(): assert response.text == "hello world" assert "transcripts" not in response.model_dump() + + +def test_response_mono_words_synthesize_raw_single_channel(): + """A mono (single-channel) response carries raw ``words`` but no ``transcripts``. + We synthesize a one-channel ``transcripts`` entry holding the RAW words verbatim + (``text``/``type``/spacing preserved), so callers get the same per-word tokens as + multichannel. The lossy OpenAI-format ``words`` (renamed ``word``, punctuation + dropped) stays too for OpenAI-compat consumers.""" + config = ElevenLabsAudioTranscriptionConfig() + + payload = { + "text": "hello world", + "language_code": "en", + "audio_duration_secs": 1.5, + "words": [ + {"text": "hello", "start": 0.0, "end": 0.4, "type": "word"}, + {"text": " ", "start": 0.4, "end": 0.5, "type": "spacing"}, + {"text": "world", "start": 0.5, "end": 0.9, "type": "word"}, + ], + } + + response = config.transform_audio_transcription_response(_response(payload)) + + # Synthesized single channel with the RAW words (spacing + type intact). + assert [t["channel_index"] for t in response["transcripts"]] == [0] + raw = response["transcripts"][0]["words"] + assert [w["text"] for w in raw] == ["hello", " ", "world"] + assert [w["type"] for w in raw] == ["word", "spacing", "word"] + # OpenAI-format words unchanged: only real words, renamed key, no spacing. + assert response["words"] == [ + {"word": "hello", "start": 0.0, "end": 0.4}, + {"word": "world", "start": 0.5, "end": 0.9}, + ]