fix(audio): handle plain-text response body for response_format=text

This commit is contained in:
cohml 2026-06-10 00:17:38 -04:00 • committed by Chris Hamill
parent 01556557f4
commit ae807d3325
2 changed files with 34 additions and 5 deletions

View file

@ -1,3 +1,4 @@
import json
from typing import List, Optional, Union
from httpx import Headers, Response
@ -131,10 +132,8 @@ class OpenAIWhisperAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
) -> TranscriptionResponse:
try:
raw_response_json = raw_response.json()
except Exception as e:
raise ValueError(
f"Error transforming response to json: {str(e)}\nResponse: {raw_response.text}"
)
except json.JSONDecodeError:
return TranscriptionResponse(text=raw_response.text)
if any(
key in raw_response_json

View file

@ -1,8 +1,11 @@
"""
Tests for OpenAIWhisperAudioTranscriptionConfig.transform_audio_transcription_request.
Tests for OpenAIWhisperAudioTranscriptionConfig.transform_audio_transcription_request
and transform_audio_transcription_response.
"""
import io
import json
from unittest.mock import MagicMock
from litellm.llms.openai.transcriptions.whisper_transformation import (
OpenAIWhisperAudioTranscriptionConfig,
@ -41,3 +44,30 @@ class TestWhisperTransformRequestResponseFormat:
"""verbose_json explicitly set by the caller stays as-is."""
data = self._transform({"response_format": "verbose_json"})
assert data["response_format"] == "verbose_json"
class TestWhisperTransformResponse:
def _make_response(self, *, text: str, is_json: bool):
mock = MagicMock()
if is_json:
mock.json.return_value = {"text": text}
else:
mock.json.side_effect = json.JSONDecodeError("", "", 0)
mock.text = text
return mock
def test_parses_json_response(self):
"""JSON body (verbose_json or json format) is parsed into TranscriptionResponse."""
config = OpenAIWhisperAudioTranscriptionConfig()
result = config.transform_audio_transcription_response(
self._make_response(text="Hello world", is_json=True)
)
assert result.text == "Hello world"
def test_parses_plain_text_response(self):
"""Plain-text body (response_format=text) is returned as TranscriptionResponse without error."""
config = OpenAIWhisperAudioTranscriptionConfig()
result = config.transform_audio_transcription_response(
self._make_response(text="Four score and seven years ago", is_json=False)
)
assert result.text == "Four score and seven years ago"