fix(cost): bill per-second transcription models outside chat modes (#44458)

* fix(cost): bill per-second transcription models outside chat modes

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(cost): restore request-time billing for per-second transcription models

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(cost): bill audio length for per-second transcription models when known

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-05 13:36:29 -07:00 • committed by GitHub
parent 9cdedf81cd
commit cad87a900f
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 71 additions and 3 deletions

View file

@ -326,7 +326,7 @@ class OCRPricing(TypedDict, total=False):
annotation_cost_per_page: ReadOnly[float | None]
_WALL_CLOCK_PRICED_MODES: Final = frozenset({"chat", "completion", "embedding", "responses"})
_WALL_CLOCK_PRICED_MODES: Final = frozenset({"audio_transcription", "chat", "completion", "embedding", "responses"})
def _has_token_or_tiered_pricing(model_info: ModelInfoBase) -> bool:
@ -346,6 +346,7 @@ def _per_second_pricing_cost(
model: str,
custom_llm_provider: str | None,
response_time_ms: float | None,
audio_seconds: float = 0.0,
) -> tuple[float, float] | None:
try:
model_info: Final = _cached_get_model_info_helper(model=model, custom_llm_provider=custom_llm_provider)
@ -366,7 +367,11 @@ def _per_second_pricing_cost(
if resolved_cost_per_second is None:
return None
seconds: Final = (response_time_ms or 0.0) / 1000
seconds: Final = (
audio_seconds
if audio_seconds > 0 and model_info.get("mode") == "audio_transcription"
else (response_time_ms or 0.0) / 1000
)
verbose_logger.debug(
"For model=%s - cost_per_second: %s; response time: %s",
model,
@ -667,6 +672,7 @@ def cost_per_token(
model=model,
custom_llm_provider=custom_llm_provider,
response_time_ms=response_time_ms,
audio_seconds=audio_transcription_file_duration,
)
) is not None:
return per_second_cost

View file

@ -4,6 +4,7 @@ from collections.abc import Mapping
from typing import Final
import httpx
import litellm
import pytest
from pydantic import JsonValue
@ -197,3 +198,41 @@ def test_streaming_chat_per_second_pricing_covers_the_full_stream(
f"spend={spend}, total frame delay={total_frame_delay_seconds}s, body={body}"
)
assert not PRICING_FIELDS.intersection(body), body
@pytest.mark.parametrize(
("model", "custom_llm_provider", "expected_cost"),
(
("deepgram/nova-3", "deepgram", (0.0007167, 0.0)),
("mistral/voxtral-mini-transcribe-realtime-latest", "mistral", (0.001, 0.0)),
),
ids=("deepgram_nova_3", "mistral_voxtral_realtime"),
)
def test_transcription_per_second_model_bills_request_time_through_cost_per_token(
model: str, custom_llm_provider: str, expected_cost: tuple[float, float]
) -> None:
cost: Final = litellm.cost_per_token(model=model, custom_llm_provider=custom_llm_provider, response_time_ms=10_000.0)
assert cost == pytest.approx(expected_cost)
def test_transcription_per_second_model_bills_audio_length_through_cost_per_token() -> None:
cost: Final = litellm.cost_per_token(
model="deepgram/nova-3",
custom_llm_provider="deepgram",
response_time_ms=10_000.0,
audio_transcription_file_duration=60.0,
)
assert cost == pytest.approx((0.0043002, 0.0))
def test_transcription_response_still_bills_audio_length_through_completion_cost() -> None:
response: Final = litellm.TranscriptionResponse(text="hello")
response.duration = 60.0
cost: Final = litellm.completion_cost(
completion_response=response, model="deepgram/nova-3", custom_llm_provider="deepgram", total_time=10.0
)
assert cost == pytest.approx(0.0043002)

View file

@ -3299,7 +3299,7 @@ def test_completion_cost_per_second_deployment_bills_the_call_duration(
assert cost == pytest.approx(0.02 * expected_seconds)
@pytest.mark.parametrize("mode", ["audio_transcription", "audio_speech", "video_generation", "realtime"])
@pytest.mark.parametrize("mode", ["audio_speech", "video_generation", "realtime"])
def test_cost_per_token_leaves_media_second_rates_to_their_dedicated_paths(monkeypatch, mode: str):
"""
A media-mode entry's per-second rates price audio or video seconds, which the dedicated
@ -3316,6 +3316,29 @@ def test_cost_per_token_leaves_media_second_rates_to_their_dedicated_paths(monke
assert cost_per_token(model=model, custom_llm_provider="openai", response_time_ms=2000.0) == (0.0, 0.0)
@pytest.mark.parametrize(
("audio_seconds", "expected_cost"),
[(0.0, (0.04, 0.0)), (60.0, (1.2, 0.0))],
ids=["no_audio_length_bills_request_time", "audio_length_bills_audio_seconds"],
)
def test_cost_per_token_bills_transcription_second_rates(
monkeypatch: pytest.MonkeyPatch, audio_seconds: float, expected_cost: tuple[float, float]
) -> None:
model: Final = "test-transcription-per-second"
monkeypatch.setitem(
litellm.model_cost,
model,
{"input_cost_per_second": 0.02, "litellm_provider": "deepgram", "mode": "audio_transcription"},
)
assert cost_per_token(
model=model,
custom_llm_provider="deepgram",
response_time_ms=2000.0,
audio_transcription_file_duration=audio_seconds,
) == pytest.approx(expected_cost)
def test_completion_cost_video_status_poll_bills_nothing_on_a_per_second_video_model(monkeypatch):
"""
Polling a video job returns a ``VideoObject`` with no stamped duration, so the cost path falls