From 01556557f4e40805866fa166ed1580241ea35d6d Mon Sep 17 00:00:00 2001 From: cohml <62400541+cohml@users.noreply.github.com> Date: Tue, 9 Jun 2026 22:34:23 -0400 Subject: [PATCH] fix(audio): don't override explicit response_format with verbose_json --- .../transcriptions/whisper_transformation.py | 4 +- .../test_whisper_transformation.py | 43 +++++++++++++++++++ 2 files changed, 44 insertions(+), 3 deletions(-) create mode 100644 tests/test_litellm/llms/openai/transcriptions/test_whisper_transformation.py diff --git a/litellm/llms/openai/transcriptions/whisper_transformation.py b/litellm/llms/openai/transcriptions/whisper_transformation.py index fa507e1bc26..170b90a3bdb 100644 --- a/litellm/llms/openai/transcriptions/whisper_transformation.py +++ b/litellm/llms/openai/transcriptions/whisper_transformation.py @@ -107,9 +107,7 @@ class OpenAIWhisperAudioTranscriptionConfig(BaseAudioTranscriptionConfig): """ data = {"model": model, "file": audio_file, **optional_params} - if "response_format" not in data or ( - data["response_format"] == "text" or data["response_format"] == "json" - ): + if "response_format" not in data: data["response_format"] = ( "verbose_json" # ensures 'duration' is received - used for cost calculation ) diff --git a/tests/test_litellm/llms/openai/transcriptions/test_whisper_transformation.py b/tests/test_litellm/llms/openai/transcriptions/test_whisper_transformation.py new file mode 100644 index 00000000000..808e29a2d6f --- /dev/null +++ b/tests/test_litellm/llms/openai/transcriptions/test_whisper_transformation.py @@ -0,0 +1,43 @@ +""" +Tests for OpenAIWhisperAudioTranscriptionConfig.transform_audio_transcription_request. +""" + +import io + +from litellm.llms.openai.transcriptions.whisper_transformation import ( + OpenAIWhisperAudioTranscriptionConfig, +) + + +class TestWhisperTransformRequestResponseFormat: + def _transform(self, optional_params: dict) -> dict: + config = OpenAIWhisperAudioTranscriptionConfig() + audio_file = io.BytesIO(b"fake audio") + audio_file.name = "test.wav" + result = config.transform_audio_transcription_request( + model="whisper-1", + audio_file=audio_file, + optional_params=optional_params, + litellm_params={}, + ) + return result.data + + def test_defaults_to_verbose_json_when_unset(self): + """When response_format is not specified, default to verbose_json for cost calculation.""" + data = self._transform({}) + assert data["response_format"] == "verbose_json" + + def test_respects_explicit_json(self): + """When response_format='json' is set, do not override to verbose_json.""" + data = self._transform({"response_format": "json"}) + assert data["response_format"] == "json" + + def test_respects_explicit_text(self): + """When response_format='text' is set, do not override to verbose_json.""" + data = self._transform({"response_format": "text"}) + assert data["response_format"] == "text" + + def test_preserves_verbose_json_when_set(self): + """verbose_json explicitly set by the caller stays as-is.""" + data = self._transform({"response_format": "verbose_json"}) + assert data["response_format"] == "verbose_json"