diff --git a/litellm/llms/openai/transcriptions/whisper_transformation.py b/litellm/llms/openai/transcriptions/whisper_transformation.py index 667e183f934..dc46918874b 100644 --- a/litellm/llms/openai/transcriptions/whisper_transformation.py +++ b/litellm/llms/openai/transcriptions/whisper_transformation.py @@ -133,6 +133,9 @@ class OpenAIWhisperAudioTranscriptionConfig(BaseAudioTranscriptionConfig): try: raw_response_json = raw_response.json() except json.JSONDecodeError: + content_type = raw_response.headers.get("content-type", "") + if "application/json" in content_type: + raise return TranscriptionResponse(text=raw_response.text) if any( diff --git a/tests/test_litellm/llms/openai/transcriptions/test_whisper_transformation.py b/tests/test_litellm/llms/openai/transcriptions/test_whisper_transformation.py index 0455fd844f3..2c34c13e64e 100644 --- a/tests/test_litellm/llms/openai/transcriptions/test_whisper_transformation.py +++ b/tests/test_litellm/llms/openai/transcriptions/test_whisper_transformation.py @@ -7,6 +7,8 @@ import io import json from unittest.mock import MagicMock +import pytest + from litellm.llms.openai.transcriptions.whisper_transformation import ( OpenAIWhisperAudioTranscriptionConfig, ) @@ -47,8 +49,9 @@ class TestWhisperTransformRequestResponseFormat: class TestWhisperTransformResponse: - def _make_response(self, *, text: str, is_json: bool): + def _make_response(self, *, text: str, content_type: str, is_json: bool): mock = MagicMock() + mock.headers = {"content-type": content_type} if is_json: mock.json.return_value = {"text": text} else: @@ -60,7 +63,9 @@ class TestWhisperTransformResponse: """JSON body (verbose_json or json format) is parsed into TranscriptionResponse.""" config = OpenAIWhisperAudioTranscriptionConfig() result = config.transform_audio_transcription_response( - self._make_response(text="Hello world", is_json=True) + self._make_response( + text="Hello world", content_type="application/json", is_json=True + ) ) assert result.text == "Hello world" @@ -68,6 +73,22 @@ class TestWhisperTransformResponse: """Plain-text body (response_format=text) is returned as TranscriptionResponse without error.""" config = OpenAIWhisperAudioTranscriptionConfig() result = config.transform_audio_transcription_response( - self._make_response(text="Four score and seven years ago", is_json=False) + self._make_response( + text="Four score and seven years ago", + content_type="text/plain", + is_json=False, + ) ) assert result.text == "Four score and seven years ago" + + def test_malformed_json_body_with_json_content_type_raises(self): + """A non-JSON body labelled application/json is a genuine upstream error, not a transcription.""" + config = OpenAIWhisperAudioTranscriptionConfig() + with pytest.raises(json.JSONDecodeError): + config.transform_audio_transcription_response( + self._make_response( + text="502 Bad Gateway", + content_type="application/json", + is_json=False, + ) + )