fix(elevenlabs/stt): synthesize single-channel transcripts for mono responses

A mono Scribe response has raw words but no transcripts array; the OpenAI-format
words we map are lossy (text->word, spacing/punctuation dropped), so callers that
segment bubbles from raw per-word tokens got an empty transcript for mono audio.
Wrap the raw words as a one-channel transcripts entry. Non-breaking: multichannel
and OpenAI-compat words unchanged.
This commit is contained in:
Tech Carrement 2026-07-06 22:06:31 +02:00
parent 22e3acc458
commit 4b73d62f4e
2 changed files with 43 additions and 0 deletions

View file

@ -162,6 +162,16 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
transcripts = response_json.get("transcripts")
if isinstance(transcripts, list):
response["transcripts"] = transcripts
elif isinstance(response_json.get("words"), list):
# Single-channel (mono) response: no `transcripts` array, only flat
# top-level `words`. The OpenAI-format `response["words"]` above is
# lossy — it renames `text`->`word` and drops spacing/punctuation and
# audio events. Wrap the RAW words as one channel so callers get the
# same verbatim per-word tokens as multichannel and can segment
# bubbles uniformly; without this a mono transcript loses punctuation.
response["transcripts"] = [
{"channel_index": 0, "words": response_json["words"]}
]
# Carry the billed audio duration (present on both single- and
# multi-channel responses) so callers can attribute cost.

View file

@ -75,3 +75,36 @@ def test_response_single_channel_is_unchanged():
assert response.text == "hello world"
assert "transcripts" not in response.model_dump()
def test_response_mono_words_synthesize_raw_single_channel():
"""A mono (single-channel) response carries raw ``words`` but no ``transcripts``.
We synthesize a one-channel ``transcripts`` entry holding the RAW words verbatim
(``text``/``type``/spacing preserved), so callers get the same per-word tokens as
multichannel. The lossy OpenAI-format ``words`` (renamed ``word``, punctuation
dropped) stays too for OpenAI-compat consumers."""
config = ElevenLabsAudioTranscriptionConfig()
payload = {
"text": "hello world",
"language_code": "en",
"audio_duration_secs": 1.5,
"words": [
{"text": "hello", "start": 0.0, "end": 0.4, "type": "word"},
{"text": " ", "start": 0.4, "end": 0.5, "type": "spacing"},
{"text": "world", "start": 0.5, "end": 0.9, "type": "word"},
],
}
response = config.transform_audio_transcription_response(_response(payload))
# Synthesized single channel with the RAW words (spacing + type intact).
assert [t["channel_index"] for t in response["transcripts"]] == [0]
raw = response["transcripts"][0]["words"]
assert [w["text"] for w in raw] == ["hello", " ", "world"]
assert [w["type"] for w in raw] == ["word", "spacing", "word"]
# OpenAI-format words unchanged: only real words, renamed key, no spacing.
assert response["words"] == [
{"word": "hello", "start": 0.0, "end": 0.4},
{"word": "world", "start": 0.5, "end": 0.9},
]