mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cost): bill per-second transcription models outside chat modes (#44458)
* fix(cost): bill per-second transcription models outside chat modes Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost): restore request-time billing for per-second transcription models Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost): bill audio length for per-second transcription models when known Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
9cdedf81cd
commit
cad87a900f
3 changed files with 71 additions and 3 deletions
|
|
@ -326,7 +326,7 @@ class OCRPricing(TypedDict, total=False):
|
|||
annotation_cost_per_page: ReadOnly[float | None]
|
||||
|
||||
|
||||
_WALL_CLOCK_PRICED_MODES: Final = frozenset({"chat", "completion", "embedding", "responses"})
|
||||
_WALL_CLOCK_PRICED_MODES: Final = frozenset({"audio_transcription", "chat", "completion", "embedding", "responses"})
|
||||
|
||||
|
||||
def _has_token_or_tiered_pricing(model_info: ModelInfoBase) -> bool:
|
||||
|
|
@ -346,6 +346,7 @@ def _per_second_pricing_cost(
|
|||
model: str,
|
||||
custom_llm_provider: str | None,
|
||||
response_time_ms: float | None,
|
||||
audio_seconds: float = 0.0,
|
||||
) -> tuple[float, float] | None:
|
||||
try:
|
||||
model_info: Final = _cached_get_model_info_helper(model=model, custom_llm_provider=custom_llm_provider)
|
||||
|
|
@ -366,7 +367,11 @@ def _per_second_pricing_cost(
|
|||
if resolved_cost_per_second is None:
|
||||
return None
|
||||
|
||||
seconds: Final = (response_time_ms or 0.0) / 1000
|
||||
seconds: Final = (
|
||||
audio_seconds
|
||||
if audio_seconds > 0 and model_info.get("mode") == "audio_transcription"
|
||||
else (response_time_ms or 0.0) / 1000
|
||||
)
|
||||
verbose_logger.debug(
|
||||
"For model=%s - cost_per_second: %s; response time: %s",
|
||||
model,
|
||||
|
|
@ -667,6 +672,7 @@ def cost_per_token(
|
|||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
response_time_ms=response_time_ms,
|
||||
audio_seconds=audio_transcription_file_duration,
|
||||
)
|
||||
) is not None:
|
||||
return per_second_cost
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ from collections.abc import Mapping
|
|||
from typing import Final
|
||||
|
||||
import httpx
|
||||
import litellm
|
||||
import pytest
|
||||
from pydantic import JsonValue
|
||||
|
||||
|
|
@ -197,3 +198,41 @@ def test_streaming_chat_per_second_pricing_covers_the_full_stream(
|
|||
f"spend={spend}, total frame delay={total_frame_delay_seconds}s, body={body}"
|
||||
)
|
||||
assert not PRICING_FIELDS.intersection(body), body
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "custom_llm_provider", "expected_cost"),
|
||||
(
|
||||
("deepgram/nova-3", "deepgram", (0.0007167, 0.0)),
|
||||
("mistral/voxtral-mini-transcribe-realtime-latest", "mistral", (0.001, 0.0)),
|
||||
),
|
||||
ids=("deepgram_nova_3", "mistral_voxtral_realtime"),
|
||||
)
|
||||
def test_transcription_per_second_model_bills_request_time_through_cost_per_token(
|
||||
model: str, custom_llm_provider: str, expected_cost: tuple[float, float]
|
||||
) -> None:
|
||||
cost: Final = litellm.cost_per_token(model=model, custom_llm_provider=custom_llm_provider, response_time_ms=10_000.0)
|
||||
|
||||
assert cost == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
def test_transcription_per_second_model_bills_audio_length_through_cost_per_token() -> None:
|
||||
cost: Final = litellm.cost_per_token(
|
||||
model="deepgram/nova-3",
|
||||
custom_llm_provider="deepgram",
|
||||
response_time_ms=10_000.0,
|
||||
audio_transcription_file_duration=60.0,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx((0.0043002, 0.0))
|
||||
|
||||
|
||||
def test_transcription_response_still_bills_audio_length_through_completion_cost() -> None:
|
||||
response: Final = litellm.TranscriptionResponse(text="hello")
|
||||
response.duration = 60.0
|
||||
|
||||
cost: Final = litellm.completion_cost(
|
||||
completion_response=response, model="deepgram/nova-3", custom_llm_provider="deepgram", total_time=10.0
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(0.0043002)
|
||||
|
|
|
|||
|
|
@ -3299,7 +3299,7 @@ def test_completion_cost_per_second_deployment_bills_the_call_duration(
|
|||
assert cost == pytest.approx(0.02 * expected_seconds)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("mode", ["audio_transcription", "audio_speech", "video_generation", "realtime"])
|
||||
@pytest.mark.parametrize("mode", ["audio_speech", "video_generation", "realtime"])
|
||||
def test_cost_per_token_leaves_media_second_rates_to_their_dedicated_paths(monkeypatch, mode: str):
|
||||
"""
|
||||
A media-mode entry's per-second rates price audio or video seconds, which the dedicated
|
||||
|
|
@ -3316,6 +3316,29 @@ def test_cost_per_token_leaves_media_second_rates_to_their_dedicated_paths(monke
|
|||
assert cost_per_token(model=model, custom_llm_provider="openai", response_time_ms=2000.0) == (0.0, 0.0)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("audio_seconds", "expected_cost"),
|
||||
[(0.0, (0.04, 0.0)), (60.0, (1.2, 0.0))],
|
||||
ids=["no_audio_length_bills_request_time", "audio_length_bills_audio_seconds"],
|
||||
)
|
||||
def test_cost_per_token_bills_transcription_second_rates(
|
||||
monkeypatch: pytest.MonkeyPatch, audio_seconds: float, expected_cost: tuple[float, float]
|
||||
) -> None:
|
||||
model: Final = "test-transcription-per-second"
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
model,
|
||||
{"input_cost_per_second": 0.02, "litellm_provider": "deepgram", "mode": "audio_transcription"},
|
||||
)
|
||||
|
||||
assert cost_per_token(
|
||||
model=model,
|
||||
custom_llm_provider="deepgram",
|
||||
response_time_ms=2000.0,
|
||||
audio_transcription_file_duration=audio_seconds,
|
||||
) == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
def test_completion_cost_video_status_poll_bills_nothing_on_a_per_second_video_model(monkeypatch):
|
||||
"""
|
||||
Polling a video job returns a ``VideoObject`` with no stamped duration, so the cost path falls
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue