mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(transcription): stop a zero output rate from zeroing transcription cost (#36914)
cost_per_second treated a declared-but-zero output_cost_per_second as a real rate, so the output branch claimed the call and the elif locked out input_cost_per_second. Every transcription model shipping output_cost_per_second 0.0 next to a real input rate billed $0, which covers 43 of the 55 per-second entries in the cost map: all 36 deepgram models, both assemblyai, both elevenlabs scribe, both groq whisper and azure-stt. Custom deployments pairing the two fields the same way billed $0 as well Take the output branch only when that rate is actually billable, so a zero falls through to the input rate. Entries that duplicate one rate into both fields, whisper-1 among them, keep billing exactly what they bill today
This commit is contained in:
parent
865ed96765
commit
29fe342ead
2 changed files with 87 additions and 3 deletions
|
|
@ -109,15 +109,16 @@ def cost_per_second(model: str, custom_llm_provider: str | None, duration: float
|
|||
prompt_cost = 0.0
|
||||
completion_cost = 0.0
|
||||
## Speech / Audio cost calculation
|
||||
if "output_cost_per_second" in model_info and model_info["output_cost_per_second"] is not None:
|
||||
output_cost_per_second: Final = model_info.get("output_cost_per_second")
|
||||
if output_cost_per_second is not None and output_cost_per_second > 0:
|
||||
verbose_logger.debug(
|
||||
"For model=%s - output_cost_per_second: %s; duration: %s",
|
||||
model,
|
||||
model_info.get("output_cost_per_second"),
|
||||
output_cost_per_second,
|
||||
duration,
|
||||
)
|
||||
## COST PER SECOND ##
|
||||
completion_cost = model_info["output_cost_per_second"] * duration
|
||||
completion_cost = output_cost_per_second * duration
|
||||
elif "input_cost_per_second" in model_info and model_info["input_cost_per_second"] is not None:
|
||||
verbose_logger.debug(
|
||||
"For model=%s - input_cost_per_second: %s; duration: %s",
|
||||
|
|
|
|||
83
tests/test_litellm/llms/openai/test_cost_calculation.py
Normal file
83
tests/test_litellm/llms/openai/test_cost_calculation.py
Normal file
|
|
@ -0,0 +1,83 @@
|
|||
"""Tests for per-second transcription cost calculation."""
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.openai.cost_calculation import cost_per_second
|
||||
|
||||
|
||||
def _register_stt(name: str, **pricing: float) -> None:
|
||||
litellm.register_model(
|
||||
{
|
||||
name: {
|
||||
"mode": "audio_transcription",
|
||||
"litellm_provider": "openai",
|
||||
**pricing,
|
||||
}
|
||||
},
|
||||
persist_across_reloads=False,
|
||||
)
|
||||
|
||||
|
||||
def test_input_rate_bills_when_output_rate_is_zero():
|
||||
"""A declared-but-zero output rate must not suppress the real input rate."""
|
||||
_register_stt(
|
||||
"test-stt-zero-output",
|
||||
input_cost_per_second=5e-05,
|
||||
output_cost_per_second=0.0,
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_second(
|
||||
model="test-stt-zero-output", custom_llm_provider="openai", duration=300.0
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(0.015)
|
||||
assert completion_cost == 0.0
|
||||
|
||||
|
||||
def test_output_rate_takes_precedence_when_both_are_billable():
|
||||
"""Entries duplicating one rate into both fields must not be billed twice."""
|
||||
_register_stt(
|
||||
"test-stt-both-rates",
|
||||
input_cost_per_second=1e-04,
|
||||
output_cost_per_second=1e-04,
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_second(
|
||||
model="test-stt-both-rates", custom_llm_provider="openai", duration=10.0
|
||||
)
|
||||
|
||||
assert prompt_cost + completion_cost == pytest.approx(1e-03)
|
||||
|
||||
|
||||
def test_output_rate_alone_still_bills():
|
||||
_register_stt("test-stt-output-only", output_cost_per_second=3e-05)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_second(
|
||||
model="test-stt-output-only", custom_llm_provider="openai", duration=60.0
|
||||
)
|
||||
|
||||
assert prompt_cost == 0.0
|
||||
assert completion_cost == pytest.approx(1.8e-03)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, provider",
|
||||
[
|
||||
("deepgram/nova-3", "deepgram"),
|
||||
("groq/whisper-large-v3", "groq"),
|
||||
("elevenlabs/scribe_v1", "elevenlabs"),
|
||||
("assemblyai/best", "assemblyai"),
|
||||
("whisper-1", "openai"),
|
||||
],
|
||||
)
|
||||
def test_shipped_per_second_models_bill_a_non_zero_cost(model, provider):
|
||||
prompt_cost, completion_cost = cost_per_second(model=model, custom_llm_provider=provider, duration=60.0)
|
||||
|
||||
assert prompt_cost + completion_cost > 0.0
|
||||
|
||||
|
||||
def test_whisper_bills_its_documented_rate_once():
|
||||
prompt_cost, completion_cost = cost_per_second(model="whisper-1", custom_llm_provider="openai", duration=30.0)
|
||||
|
||||
assert prompt_cost + completion_cost == pytest.approx(0.003)
|
||||
Loading…
Add table
Reference in a new issue