mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(elevenlabs/stt): synthesize single-channel transcripts for mono responses
A mono Scribe response has raw words but no transcripts array; the OpenAI-format words we map are lossy (text->word, spacing/punctuation dropped), so callers that segment bubbles from raw per-word tokens got an empty transcript for mono audio. Wrap the raw words as a one-channel transcripts entry. Non-breaking: multichannel and OpenAI-compat words unchanged.
This commit is contained in:
parent
22e3acc458
commit
4b73d62f4e
2 changed files with 43 additions and 0 deletions
|
|
@ -162,6 +162,16 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
transcripts = response_json.get("transcripts")
|
||||
if isinstance(transcripts, list):
|
||||
response["transcripts"] = transcripts
|
||||
elif isinstance(response_json.get("words"), list):
|
||||
# Single-channel (mono) response: no `transcripts` array, only flat
|
||||
# top-level `words`. The OpenAI-format `response["words"]` above is
|
||||
# lossy — it renames `text`->`word` and drops spacing/punctuation and
|
||||
# audio events. Wrap the RAW words as one channel so callers get the
|
||||
# same verbatim per-word tokens as multichannel and can segment
|
||||
# bubbles uniformly; without this a mono transcript loses punctuation.
|
||||
response["transcripts"] = [
|
||||
{"channel_index": 0, "words": response_json["words"]}
|
||||
]
|
||||
|
||||
# Carry the billed audio duration (present on both single- and
|
||||
# multi-channel responses) so callers can attribute cost.
|
||||
|
|
|
|||
|
|
@ -75,3 +75,36 @@ def test_response_single_channel_is_unchanged():
|
|||
|
||||
assert response.text == "hello world"
|
||||
assert "transcripts" not in response.model_dump()
|
||||
|
||||
|
||||
def test_response_mono_words_synthesize_raw_single_channel():
|
||||
"""A mono (single-channel) response carries raw ``words`` but no ``transcripts``.
|
||||
We synthesize a one-channel ``transcripts`` entry holding the RAW words verbatim
|
||||
(``text``/``type``/spacing preserved), so callers get the same per-word tokens as
|
||||
multichannel. The lossy OpenAI-format ``words`` (renamed ``word``, punctuation
|
||||
dropped) stays too for OpenAI-compat consumers."""
|
||||
config = ElevenLabsAudioTranscriptionConfig()
|
||||
|
||||
payload = {
|
||||
"text": "hello world",
|
||||
"language_code": "en",
|
||||
"audio_duration_secs": 1.5,
|
||||
"words": [
|
||||
{"text": "hello", "start": 0.0, "end": 0.4, "type": "word"},
|
||||
{"text": " ", "start": 0.4, "end": 0.5, "type": "spacing"},
|
||||
{"text": "world", "start": 0.5, "end": 0.9, "type": "word"},
|
||||
],
|
||||
}
|
||||
|
||||
response = config.transform_audio_transcription_response(_response(payload))
|
||||
|
||||
# Synthesized single channel with the RAW words (spacing + type intact).
|
||||
assert [t["channel_index"] for t in response["transcripts"]] == [0]
|
||||
raw = response["transcripts"][0]["words"]
|
||||
assert [w["text"] for w in raw] == ["hello", " ", "world"]
|
||||
assert [w["type"] for w in raw] == ["word", "spacing", "word"]
|
||||
# OpenAI-format words unchanged: only real words, renamed key, no spacing.
|
||||
assert response["words"] == [
|
||||
{"word": "hello", "start": 0.0, "end": 0.4},
|
||||
{"word": "world", "start": 0.5, "end": 0.9},
|
||||
]
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue