fix(realtime): preserve batch cache rates and Azure v1 routes

This commit is contained in:
Emerson Gomes 2026-09-23 04:34:03 -05:00
parent 275bc4c524
commit 9e1d36489e
No known key found for this signature in database
GPG key ID: D3DF28AB5D1B5E17
5 changed files with 100 additions and 35 deletions

View file

@ -43,8 +43,15 @@ class AzureAudioTranscription(AzureChatCompletion):
) -> TranscriptionResponse | Coroutine[Any, Any, TranscriptionResponse]:
data: Final = {"model": model, "file": audio_file, **optional_params}
sdk_data: Final = sdk_compatible_transcription_request_data(data)
model_info: Final = litellm.model_cost.get(f"azure/{model}")
provider_specific_entry: Final = model_info.get("provider_specific_entry") if model_info is not None else None
requires_deployment_api: Final = model_info is None or (
provider_specific_entry is not None and provider_specific_entry.get("transcription_deployment_api") == 1
)
resolved_api_version: Final = (
litellm.AZURE_DEFAULT_API_VERSION if api_version in ("v1", "latest", "preview") else api_version
litellm.AZURE_DEFAULT_API_VERSION
if requires_deployment_api and api_version in ("v1", "latest", "preview")
else api_version
)
if atranscription is True:

View file

@ -234,6 +234,7 @@
"mode": "audio_transcription",
"provider_specific_entry": {
"realtime_ga_only": 1,
"transcription_deployment_api": 1,
"transcription_json_only": 1
},
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
@ -25728,6 +25729,7 @@
},
"gemini-3-pro-image": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
"cache_read_input_token_cost_flex": 1e-07,
@ -25822,6 +25824,7 @@
},
"gemini-3.1-flash-image": {
"cache_read_input_token_cost": 5e-08,
"cache_read_input_token_cost_batches": 2.5e-08,
"cache_read_input_token_cost_flex": 2.5e-08,
"deprecation_date": "2027-05-28",
"input_cost_per_image": 0.00056,
@ -25905,6 +25908,7 @@
},
"gemini-3.1-flash-lite-image": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"input_cost_per_image": 0.00028,
"input_cost_per_token": 2.5e-07,
@ -25997,6 +26001,7 @@
"cache_read_input_audio_token_cost": 5e-08,
"deprecation_date": "2027-05-07",
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"input_cost_per_audio_token": 5e-07,
@ -26056,6 +26061,7 @@
"gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_batches": 1.5e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5.4e-08,
"input_cost_per_token": 3e-07,
@ -26112,6 +26118,7 @@
},
"deep-research-pro-preview-12-2025": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"input_cost_per_image": 0.0011,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
@ -26717,6 +26724,7 @@
"prompt_cache_min_tokens": 4096,
"deprecation_date": "2027-05-19",
"cache_read_input_token_cost": 1.5e-07,
"cache_read_input_token_cost_batches": 7.5e-08,
"input_cost_per_token": 1.5e-06,
"input_cost_per_audio_token": 1.5e-06,
"litellm_provider": "vertex_ai",
@ -26776,6 +26784,7 @@
"vertex_ai/gemini-3.6-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -26833,6 +26842,7 @@
"vertex_ai/gemini-3.7-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -26891,6 +26901,7 @@
"vertex_ai/gemini-3.8-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28638,6 +28649,7 @@
"prompt_cache_min_tokens": 4096,
"deprecation_date": "2027-05-19",
"cache_read_input_token_cost": 1.5e-07,
"cache_read_input_token_cost_batches": 7.5e-08,
"input_cost_per_audio_token": 1.5e-06,
"input_cost_per_token": 1.5e-06,
"litellm_provider": "vertex_ai-language-models",
@ -28697,6 +28709,7 @@
"gemini-3.6-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28754,6 +28767,7 @@
"gemini-3.7-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28812,6 +28826,7 @@
"gemini-3.8-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -33499,6 +33514,7 @@
},
"gpt-5-2025-08-07": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_batches": 6.25e-08,
"cache_read_input_token_cost_flex": 6.25e-08,
"cache_read_input_token_cost_priority": 2.5e-07,
"deprecation_date": "2026-12-11",
@ -33679,6 +33695,7 @@
},
"gpt-5-mini-2025-08-07": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"deprecation_date": "2026-12-11",
@ -33779,6 +33796,7 @@
},
"gpt-5-nano-2025-08-07": {
"cache_read_input_token_cost": 5e-09,
"cache_read_input_token_cost_batches": 2.5e-09,
"cache_read_input_token_cost_flex": 2.5e-09,
"deprecation_date": "2026-12-11",
"input_cost_per_token": 5e-08,
@ -48117,6 +48135,7 @@
},
"vertex_ai/gemini-3-pro-image": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
"cache_read_input_token_cost_flex": 1e-07,
@ -48163,6 +48182,7 @@
},
"vertex_ai/gemini-3.1-flash-image": {
"cache_read_input_token_cost": 5e-08,
"cache_read_input_token_cost_batches": 2.5e-08,
"cache_read_input_token_cost_flex": 2.5e-08,
"deprecation_date": "2027-05-28",
"input_cost_per_image": 0.00056,
@ -48198,6 +48218,7 @@
},
"vertex_ai/gemini-3.1-flash-lite-image": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"input_cost_per_image": 0.00028,
"input_cost_per_token": 2.5e-07,
@ -48290,6 +48311,7 @@
"cache_read_input_audio_token_cost": 5e-08,
"deprecation_date": "2027-05-07",
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"input_cost_per_audio_token": 5e-07,
@ -48350,6 +48372,7 @@
"vertex_ai/gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_batches": 1.5e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5.4e-08,
"input_cost_per_token": 3e-07,
@ -48407,6 +48430,7 @@
},
"vertex_ai/deep-research-pro-preview-12-2025": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"input_cost_per_image": 0.0011,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,

View file

@ -234,6 +234,7 @@
"mode": "audio_transcription",
"provider_specific_entry": {
"realtime_ga_only": 1,
"transcription_deployment_api": 1,
"transcription_json_only": 1
},
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
@ -25728,6 +25729,7 @@
},
"gemini-3-pro-image": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
"cache_read_input_token_cost_flex": 1e-07,
@ -25822,6 +25824,7 @@
},
"gemini-3.1-flash-image": {
"cache_read_input_token_cost": 5e-08,
"cache_read_input_token_cost_batches": 2.5e-08,
"cache_read_input_token_cost_flex": 2.5e-08,
"deprecation_date": "2027-05-28",
"input_cost_per_image": 0.00056,
@ -25905,6 +25908,7 @@
},
"gemini-3.1-flash-lite-image": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"input_cost_per_image": 0.00028,
"input_cost_per_token": 2.5e-07,
@ -25997,6 +26001,7 @@
"cache_read_input_audio_token_cost": 5e-08,
"deprecation_date": "2027-05-07",
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"input_cost_per_audio_token": 5e-07,
@ -26056,6 +26061,7 @@
"gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_batches": 1.5e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5.4e-08,
"input_cost_per_token": 3e-07,
@ -26112,6 +26118,7 @@
},
"deep-research-pro-preview-12-2025": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"input_cost_per_image": 0.0011,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
@ -26717,6 +26724,7 @@
"prompt_cache_min_tokens": 4096,
"deprecation_date": "2027-05-19",
"cache_read_input_token_cost": 1.5e-07,
"cache_read_input_token_cost_batches": 7.5e-08,
"input_cost_per_token": 1.5e-06,
"input_cost_per_audio_token": 1.5e-06,
"litellm_provider": "vertex_ai",
@ -26776,6 +26784,7 @@
"vertex_ai/gemini-3.6-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -26833,6 +26842,7 @@
"vertex_ai/gemini-3.7-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -26891,6 +26901,7 @@
"vertex_ai/gemini-3.8-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28638,6 +28649,7 @@
"prompt_cache_min_tokens": 4096,
"deprecation_date": "2027-05-19",
"cache_read_input_token_cost": 1.5e-07,
"cache_read_input_token_cost_batches": 7.5e-08,
"input_cost_per_audio_token": 1.5e-06,
"input_cost_per_token": 1.5e-06,
"litellm_provider": "vertex_ai-language-models",
@ -28697,6 +28709,7 @@
"gemini-3.6-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28754,6 +28767,7 @@
"gemini-3.7-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28812,6 +28826,7 @@
"gemini-3.8-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -33499,6 +33514,7 @@
},
"gpt-5-2025-08-07": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_batches": 6.25e-08,
"cache_read_input_token_cost_flex": 6.25e-08,
"cache_read_input_token_cost_priority": 2.5e-07,
"deprecation_date": "2026-12-11",
@ -33679,6 +33695,7 @@
},
"gpt-5-mini-2025-08-07": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"deprecation_date": "2026-12-11",
@ -33779,6 +33796,7 @@
},
"gpt-5-nano-2025-08-07": {
"cache_read_input_token_cost": 5e-09,
"cache_read_input_token_cost_batches": 2.5e-09,
"cache_read_input_token_cost_flex": 2.5e-09,
"deprecation_date": "2026-12-11",
"input_cost_per_token": 5e-08,
@ -48117,6 +48135,7 @@
},
"vertex_ai/gemini-3-pro-image": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
"cache_read_input_token_cost_flex": 1e-07,
@ -48163,6 +48182,7 @@
},
"vertex_ai/gemini-3.1-flash-image": {
"cache_read_input_token_cost": 5e-08,
"cache_read_input_token_cost_batches": 2.5e-08,
"cache_read_input_token_cost_flex": 2.5e-08,
"deprecation_date": "2027-05-28",
"input_cost_per_image": 0.00056,
@ -48198,6 +48218,7 @@
},
"vertex_ai/gemini-3.1-flash-lite-image": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"input_cost_per_image": 0.00028,
"input_cost_per_token": 2.5e-07,
@ -48290,6 +48311,7 @@
"cache_read_input_audio_token_cost": 5e-08,
"deprecation_date": "2027-05-07",
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"input_cost_per_audio_token": 5e-07,
@ -48350,6 +48372,7 @@
"vertex_ai/gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_batches": 1.5e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5.4e-08,
"input_cost_per_token": 3e-07,
@ -48407,6 +48430,7 @@
},
"vertex_ai/deep-research-pro-preview-12-2025": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"input_cost_per_image": 0.0011,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,

View file

@ -1,6 +1,8 @@
import io
import json
from pathlib import Path
from typing import Final
from unittest.mock import MagicMock
import httpx
import pytest
@ -9,6 +11,8 @@ from openai import AzureOpenAI
import litellm
from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.audio_utils.utils import calculate_request_duration
from litellm.llms.azure.audio_transcriptions import AzureAudioTranscription
from litellm.types.utils import TranscriptionResponse
AUDIO_FILE: Final = Path(__file__).parents[3] / "gettysburg.wav"
WHISPER_COST_PER_SECOND: Final = 0.0001
@ -39,3 +43,43 @@ def test_azure_transcription_keeps_the_azure_provider():
assert response._hidden_params["custom_llm_provider"] == "azure"
assert json.loads(response.model_dump_json())["text"] == "Four score and seven years ago"
@pytest.mark.parametrize("api_version", ["v1", "latest", "preview"])
@pytest.mark.parametrize(
("model", "expected_path"),
[
("whisper-1", "/openai/v1/audio/transcriptions"),
("gpt-transcribe", "/openai/deployments/gpt-transcribe/audio/transcriptions"),
("custom-transcribe-deployment", "/openai/deployments/custom-transcribe-deployment/audio/transcriptions"),
],
)
def test_azure_transcription_alias_uses_model_route(
monkeypatch: pytest.MonkeyPatch, model: str, expected_path: str, api_version: str
) -> None:
def send_response(request: httpx.Request) -> httpx.Response:
assert request.url.path == expected_path
assert request.url.params.get("api-version") == (
None if expected_path.startswith("/openai/v1/") else litellm.AZURE_DEFAULT_API_VERSION
)
return httpx.Response(200, json={"text": "hello"})
audio_file: Final = io.BytesIO(b"audio")
audio_file.name = "sample.wav"
with httpx.Client(transport=httpx.MockTransport(send_response)) as http_client:
monkeypatch.setattr(litellm, "client_session", http_client)
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
response: Final = AzureAudioTranscription().audio_transcriptions(
model=model,
audio_file=audio_file,
optional_params={"response_format": "json"},
logging_obj=MagicMock(),
model_response=TranscriptionResponse(),
timeout=10,
max_retries=0,
api_key="test-key",
api_base="https://example.openai.azure.com",
api_version=api_version,
)
assert response.text == "hello"

View file

@ -251,40 +251,6 @@ def test_gpt_live_transcribe_rejects_file_transcription(local_model_cost_map: No
)
@pytest.mark.parametrize(
("api_version", "expected_api_version"),
[
("v1", litellm.AZURE_DEFAULT_API_VERSION),
("latest", litellm.AZURE_DEFAULT_API_VERSION),
("preview", litellm.AZURE_DEFAULT_API_VERSION),
(None, None),
("2025-04-01-preview", "2025-04-01-preview"),
],
)
@pytest.mark.parametrize("model", ["gpt-transcribe", "custom-transcribe-deployment"])
def test_azure_audio_transcription_resolves_api_version_in_provider(
model: str, api_version: str | None, expected_api_version: str | None
) -> None:
handler = AzureAudioTranscription()
handler.async_audio_transcriptions = MagicMock(return_value=MagicMock())
handler.audio_transcriptions(
model=model,
audio_file=io.BytesIO(b"audio"),
optional_params={"stream": True},
logging_obj=MagicMock(),
model_response=TranscriptionResponse(),
timeout=10,
max_retries=0,
api_key="sk-test",
api_base="https://example.openai.azure.com",
api_version=api_version,
atranscription=True,
)
assert handler.async_audio_transcriptions.call_args.kwargs["api_version"] == expected_api_version
def test_azure_gpt_transcribe_uses_deployment_scoped_route():
def send_response(request: httpx.Request) -> httpx.Response:
assert str(request.url) == (