fix(transcription): stop a zero output rate from zeroing transcription cost (#36914)

cost_per_second treated a declared-but-zero output_cost_per_second as a real
rate, so the output branch claimed the call and the elif locked out
input_cost_per_second. Every transcription model shipping
output_cost_per_second 0.0 next to a real input rate billed $0, which covers
43 of the 55 per-second entries in the cost map: all 36 deepgram models, both
assemblyai, both elevenlabs scribe, both groq whisper and azure-stt. Custom
deployments pairing the two fields the same way billed $0 as well

Take the output branch only when that rate is actually billable, so a zero
falls through to the input rate. Entries that duplicate one rate into both
fields, whisper-1 among them, keep billing exactly what they bill today
This commit is contained in:
Ahmed N 2026-08-14 23:12:37 +01:00 • committed by GitHub
parent 865ed96765
commit 29fe342ead
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 87 additions and 3 deletions

View file

@ -109,15 +109,16 @@ def cost_per_second(model: str, custom_llm_provider: str | None, duration: float
prompt_cost = 0.0
completion_cost = 0.0
## Speech / Audio cost calculation
if "output_cost_per_second" in model_info and model_info["output_cost_per_second"] is not None:
output_cost_per_second: Final = model_info.get("output_cost_per_second")
if output_cost_per_second is not None and output_cost_per_second > 0:
verbose_logger.debug(
"For model=%s - output_cost_per_second: %s; duration: %s",
model,
model_info.get("output_cost_per_second"),
output_cost_per_second,
duration,
)
## COST PER SECOND ##
completion_cost = model_info["output_cost_per_second"] * duration
completion_cost = output_cost_per_second * duration
elif "input_cost_per_second" in model_info and model_info["input_cost_per_second"] is not None:
verbose_logger.debug(
"For model=%s - input_cost_per_second: %s; duration: %s",

View file

@ -0,0 +1,83 @@
"""Tests for per-second transcription cost calculation."""
import pytest
import litellm
from litellm.llms.openai.cost_calculation import cost_per_second
def _register_stt(name: str, **pricing: float) -> None:
litellm.register_model(
{
name: {
"mode": "audio_transcription",
"litellm_provider": "openai",
**pricing,
}
},
persist_across_reloads=False,
)
def test_input_rate_bills_when_output_rate_is_zero():
"""A declared-but-zero output rate must not suppress the real input rate."""
_register_stt(
"test-stt-zero-output",
input_cost_per_second=5e-05,
output_cost_per_second=0.0,
)
prompt_cost, completion_cost = cost_per_second(
model="test-stt-zero-output", custom_llm_provider="openai", duration=300.0
)
assert prompt_cost == pytest.approx(0.015)
assert completion_cost == 0.0
def test_output_rate_takes_precedence_when_both_are_billable():
"""Entries duplicating one rate into both fields must not be billed twice."""
_register_stt(
"test-stt-both-rates",
input_cost_per_second=1e-04,
output_cost_per_second=1e-04,
)
prompt_cost, completion_cost = cost_per_second(
model="test-stt-both-rates", custom_llm_provider="openai", duration=10.0
)
assert prompt_cost + completion_cost == pytest.approx(1e-03)
def test_output_rate_alone_still_bills():
_register_stt("test-stt-output-only", output_cost_per_second=3e-05)
prompt_cost, completion_cost = cost_per_second(
model="test-stt-output-only", custom_llm_provider="openai", duration=60.0
)
assert prompt_cost == 0.0
assert completion_cost == pytest.approx(1.8e-03)
@pytest.mark.parametrize(
"model, provider",
[
("deepgram/nova-3", "deepgram"),
("groq/whisper-large-v3", "groq"),
("elevenlabs/scribe_v1", "elevenlabs"),
("assemblyai/best", "assemblyai"),
("whisper-1", "openai"),
],
)
def test_shipped_per_second_models_bill_a_non_zero_cost(model, provider):
prompt_cost, completion_cost = cost_per_second(model=model, custom_llm_provider=provider, duration=60.0)
assert prompt_cost + completion_cost > 0.0
def test_whisper_bills_its_documented_rate_once():
prompt_cost, completion_cost = cost_per_second(model="whisper-1", custom_llm_provider="openai", duration=30.0)
assert prompt_cost + completion_cost == pytest.approx(0.003)