From cad87a900fbe8b99eba258e6ebb23f58225a8002 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Mon, 5 Oct 2026 13:36:29 -0700 Subject: [PATCH] fix(cost): bill per-second transcription models outside chat modes (#44458) * fix(cost): bill per-second transcription models outside chat modes Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost): restore request-time billing for per-second transcription models Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost): bill audio length for per-second transcription models when known Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/cost_calculator.py | 10 ++++- .../pricing/test_per_second_pricing.py | 39 +++++++++++++++++++ tests/unit/test_cost_calculator.py | 25 +++++++++++- 3 files changed, 71 insertions(+), 3 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index e358636c105..f3abf50bbce 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -326,7 +326,7 @@ class OCRPricing(TypedDict, total=False): annotation_cost_per_page: ReadOnly[float | None] -_WALL_CLOCK_PRICED_MODES: Final = frozenset({"chat", "completion", "embedding", "responses"}) +_WALL_CLOCK_PRICED_MODES: Final = frozenset({"audio_transcription", "chat", "completion", "embedding", "responses"}) def _has_token_or_tiered_pricing(model_info: ModelInfoBase) -> bool: @@ -346,6 +346,7 @@ def _per_second_pricing_cost( model: str, custom_llm_provider: str | None, response_time_ms: float | None, + audio_seconds: float = 0.0, ) -> tuple[float, float] | None: try: model_info: Final = _cached_get_model_info_helper(model=model, custom_llm_provider=custom_llm_provider) @@ -366,7 +367,11 @@ def _per_second_pricing_cost( if resolved_cost_per_second is None: return None - seconds: Final = (response_time_ms or 0.0) / 1000 + seconds: Final = ( + audio_seconds + if audio_seconds > 0 and model_info.get("mode") == "audio_transcription" + else (response_time_ms or 0.0) / 1000 + ) verbose_logger.debug( "For model=%s - cost_per_second: %s; response time: %s", model, @@ -667,6 +672,7 @@ def cost_per_token( model=model, custom_llm_provider=custom_llm_provider, response_time_ms=response_time_ms, + audio_seconds=audio_transcription_file_duration, ) ) is not None: return per_second_cost diff --git a/tests/integration/pricing/test_per_second_pricing.py b/tests/integration/pricing/test_per_second_pricing.py index a4958bd18c5..144d05fdcc3 100644 --- a/tests/integration/pricing/test_per_second_pricing.py +++ b/tests/integration/pricing/test_per_second_pricing.py @@ -4,6 +4,7 @@ from collections.abc import Mapping from typing import Final import httpx +import litellm import pytest from pydantic import JsonValue @@ -197,3 +198,41 @@ def test_streaming_chat_per_second_pricing_covers_the_full_stream( f"spend={spend}, total frame delay={total_frame_delay_seconds}s, body={body}" ) assert not PRICING_FIELDS.intersection(body), body + + +@pytest.mark.parametrize( + ("model", "custom_llm_provider", "expected_cost"), + ( + ("deepgram/nova-3", "deepgram", (0.0007167, 0.0)), + ("mistral/voxtral-mini-transcribe-realtime-latest", "mistral", (0.001, 0.0)), + ), + ids=("deepgram_nova_3", "mistral_voxtral_realtime"), +) +def test_transcription_per_second_model_bills_request_time_through_cost_per_token( + model: str, custom_llm_provider: str, expected_cost: tuple[float, float] +) -> None: + cost: Final = litellm.cost_per_token(model=model, custom_llm_provider=custom_llm_provider, response_time_ms=10_000.0) + + assert cost == pytest.approx(expected_cost) + + +def test_transcription_per_second_model_bills_audio_length_through_cost_per_token() -> None: + cost: Final = litellm.cost_per_token( + model="deepgram/nova-3", + custom_llm_provider="deepgram", + response_time_ms=10_000.0, + audio_transcription_file_duration=60.0, + ) + + assert cost == pytest.approx((0.0043002, 0.0)) + + +def test_transcription_response_still_bills_audio_length_through_completion_cost() -> None: + response: Final = litellm.TranscriptionResponse(text="hello") + response.duration = 60.0 + + cost: Final = litellm.completion_cost( + completion_response=response, model="deepgram/nova-3", custom_llm_provider="deepgram", total_time=10.0 + ) + + assert cost == pytest.approx(0.0043002) diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index 9a7c22f6f82..26228a36866 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -3299,7 +3299,7 @@ def test_completion_cost_per_second_deployment_bills_the_call_duration( assert cost == pytest.approx(0.02 * expected_seconds) -@pytest.mark.parametrize("mode", ["audio_transcription", "audio_speech", "video_generation", "realtime"]) +@pytest.mark.parametrize("mode", ["audio_speech", "video_generation", "realtime"]) def test_cost_per_token_leaves_media_second_rates_to_their_dedicated_paths(monkeypatch, mode: str): """ A media-mode entry's per-second rates price audio or video seconds, which the dedicated @@ -3316,6 +3316,29 @@ def test_cost_per_token_leaves_media_second_rates_to_their_dedicated_paths(monke assert cost_per_token(model=model, custom_llm_provider="openai", response_time_ms=2000.0) == (0.0, 0.0) +@pytest.mark.parametrize( + ("audio_seconds", "expected_cost"), + [(0.0, (0.04, 0.0)), (60.0, (1.2, 0.0))], + ids=["no_audio_length_bills_request_time", "audio_length_bills_audio_seconds"], +) +def test_cost_per_token_bills_transcription_second_rates( + monkeypatch: pytest.MonkeyPatch, audio_seconds: float, expected_cost: tuple[float, float] +) -> None: + model: Final = "test-transcription-per-second" + monkeypatch.setitem( + litellm.model_cost, + model, + {"input_cost_per_second": 0.02, "litellm_provider": "deepgram", "mode": "audio_transcription"}, + ) + + assert cost_per_token( + model=model, + custom_llm_provider="deepgram", + response_time_ms=2000.0, + audio_transcription_file_duration=audio_seconds, + ) == pytest.approx(expected_cost) + + def test_completion_cost_video_status_poll_bills_nothing_on_a_per_second_video_model(monkeypatch): """ Polling a video job returns a ``VideoObject`` with no stamped duration, so the cost path falls