From 8f23d203670c36386a65e5ed0fe4fe3d0be7cf6e Mon Sep 17 00:00:00 2001 From: Emerson Gomes Date: Wed, 23 Sep 2026 03:23:57 -0500 Subject: [PATCH] fix(realtime): bill retained output and route Azure aliases --- .../litellm_core_utils/realtime_streaming.py | 12 +++++++++- litellm/llms/azure/audio_transcriptions.py | 12 +--------- ...odel_prices_and_context_window_backup.json | 1 - model_prices_and_context_window.json | 1 - .../test_realtime_streaming.py | 22 +++++++++++++++++++ .../transcriptions/test_gpt_transcribe.py | 7 +++--- 6 files changed, 38 insertions(+), 17 deletions(-) diff --git a/litellm/litellm_core_utils/realtime_streaming.py b/litellm/litellm_core_utils/realtime_streaming.py index c58a7573fc0..25fc063a578 100644 --- a/litellm/litellm_core_utils/realtime_streaming.py +++ b/litellm/litellm_core_utils/realtime_streaming.py @@ -450,7 +450,17 @@ class RealTimeStreaming: output_seconds if isinstance(output_seconds, (int, float)) else synthetic_output_seconds ) if isinstance(input_seconds, (int, float)) or resolved_output_seconds is not None: - if not self._should_store_message(event_obj): + if self._should_store_message(event_obj): + if not isinstance(output_seconds, (int, float)) and synthetic_output_seconds is not None: + self.messages.append( + OpenAIRealtimeTranslationClosedEvent( + type="session.closed", + usage=OpenAIRealtimeTranslationDurationUsage( + type="duration", output_seconds=synthetic_output_seconds + ), + ) + ) + else: normalized_usage: Final = ( OpenAIRealtimeTranslationDurationUsage( type="duration", diff --git a/litellm/llms/azure/audio_transcriptions.py b/litellm/llms/azure/audio_transcriptions.py index dee8de8b7cb..754c319d0d0 100644 --- a/litellm/llms/azure/audio_transcriptions.py +++ b/litellm/llms/azure/audio_transcriptions.py @@ -43,18 +43,8 @@ class AzureAudioTranscription(AzureChatCompletion): ) -> TranscriptionResponse | Coroutine[Any, Any, TranscriptionResponse]: data: Final = {"model": model, "file": audio_file, **optional_params} sdk_data: Final = sdk_compatible_transcription_request_data(data) - model_info: Final = ( - litellm.get_model_info(model=model, custom_llm_provider="azure") - if f"azure/{model}" in litellm.model_cost - else None - ) - provider_specific_entry: Final = model_info.get("provider_specific_entry") if model_info is not None else None resolved_api_version: Final = ( - litellm.AZURE_DEFAULT_API_VERSION - if provider_specific_entry is not None - and provider_specific_entry.get("transcription_deployment_api") == 1 - and api_version in ("v1", "latest", "preview") - else api_version + litellm.AZURE_DEFAULT_API_VERSION if api_version in ("v1", "latest", "preview") else api_version ) if atranscription is True: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 50ffeb6bf99..0b2117a627e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -234,7 +234,6 @@ "mode": "audio_transcription", "provider_specific_entry": { "realtime_ga_only": 1, - "transcription_deployment_api": 1, "transcription_json_only": 1 }, "source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 50ffeb6bf99..0b2117a627e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -234,7 +234,6 @@ "mode": "audio_transcription", "provider_specific_entry": { "realtime_ga_only": 1, - "transcription_deployment_api": 1, "transcription_json_only": 1 }, "source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740", diff --git a/tests/test_litellm/litellm_core_utils/test_realtime_streaming.py b/tests/test_litellm/litellm_core_utils/test_realtime_streaming.py index 885a8e52d86..205e18e26fe 100644 --- a/tests/test_litellm/litellm_core_utils/test_realtime_streaming.py +++ b/tests/test_litellm/litellm_core_utils/test_realtime_streaming.py @@ -3027,6 +3027,28 @@ def test_translation_preserves_input_only_provider_usage( assert closed_events[0]["usage"] == expected_usage +def test_translation_retained_input_only_close_event_bills_captured_output(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "logged_real_time_event_types", "*") + streaming = RealTimeStreaming( + websocket=MagicMock(), + backend_ws=MagicMock(), + logging_obj=MagicMock(), + model="gpt-realtime-translate", + translation_session=True, + ) + streaming._translation_output_audio_bytes = 48000 + close_event: Final = {"type": "session.closed", "usage": {"type": "duration", "input_seconds": 0.25}} + + streaming._capture_translation_output_audio(close_event) + streaming.store_message(close_event) + streaming._finalize_translation_usage() + + usage_events: Final = tuple(event["usage"] for event in streaming.messages if event.get("type") == "session.closed") + assert len(usage_events) == 2 + assert sum(usage.get("input_seconds", 0.0) for usage in usage_events) == 0.25 + assert sum(usage.get("output_seconds", 0.0) for usage in usage_events) == 1.0 + + @pytest.mark.asyncio async def test_audio_delta_frame_parsed_at_most_once(): client_ws = _beta_client_ws() diff --git a/tests/unit/llms/openai/transcriptions/test_gpt_transcribe.py b/tests/unit/llms/openai/transcriptions/test_gpt_transcribe.py index 26eb6654d87..d192550943c 100644 --- a/tests/unit/llms/openai/transcriptions/test_gpt_transcribe.py +++ b/tests/unit/llms/openai/transcriptions/test_gpt_transcribe.py @@ -261,14 +261,15 @@ def test_gpt_live_transcribe_rejects_file_transcription(local_model_cost_map: No ("2025-04-01-preview", "2025-04-01-preview"), ], ) -def test_azure_gpt_transcribe_resolves_api_version_in_provider( - local_model_cost_map: None, api_version: str | None, expected_api_version: str | None +@pytest.mark.parametrize("model", ["gpt-transcribe", "custom-transcribe-deployment"]) +def test_azure_audio_transcription_resolves_api_version_in_provider( + model: str, api_version: str | None, expected_api_version: str | None ) -> None: handler = AzureAudioTranscription() handler.async_audio_transcriptions = MagicMock(return_value=MagicMock()) handler.audio_transcriptions( - model="gpt-transcribe", + model=model, audio_file=io.BytesIO(b"audio"), optional_params={"stream": True}, logging_obj=MagicMock(),