fix(transcription): accept fractional usage.seconds in diarized_json responses (#30996)

gpt-4o-transcribe and compatible ASR backends return a diarized_json
response with usage={"type": "duration", "seconds": <float>}, e.g. 295.8.
TranscriptionUsageDurationObject typed seconds as int, so parsing the
response raised a pydantic ValidationError (int_from_float). That error
surfaces as an APIConnectionError which the router treats as retryable, so
it keeps re-calling the upstream (200 every time) until the upstream
rate-limits and returns 429 to the caller.

OpenAI specs this field as a float (see openai SDK UsageDuration.seconds),
so widen seconds to float. With the parse succeeding there is no exception
left to retry, which removes the loop.

Co-authored-by: Neimar Avila <19142978+neimaravila@users.noreply.github.com>
This commit is contained in:
Neimar Avila 2026-06-24 07:43:12 -03:00 • committed by GitHub
parent e71d6ef8ba
commit 00725de9f2
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 47 additions and 2 deletions

View file

@ -2441,7 +2441,7 @@ class ImageResponse(OpenAIImageResponse, BaseLiteLLMOpenAIResponseObject):
class TranscriptionUsageDurationObject(BaseModel):
type: Literal["duration"]
seconds: int
seconds: float
class TranscriptionUsageInputTokenDetailsObject(BaseModel):

View file

@ -13,7 +13,52 @@ from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import (
convert_to_model_response_object,
)
from litellm.types.utils import TranscriptionResponse
from litellm.types.utils import (
TranscriptionResponse,
TranscriptionUsageDurationObject,
)
class TestDiarizedJsonUsageParsing:
"""gpt-4o-transcribe / diarized_json returns a fractional `usage.seconds`."""
def test_fractional_duration_seconds_does_not_raise(self):
"""
A diarized_json response carries usage={"type": "duration", "seconds": <float>}.
OpenAI specs `seconds` as a float, so a fractional value must parse cleanly
instead of raising and getting retried until the upstream rate-limits.
"""
response_object = {
"text": "speaker_1: Olá",
"task": "transcribe",
"duration": 295.8,
"segments": [
{
"id": "seg_001",
"speaker": "speaker_1",
"start": 0.0,
"end": 1.0,
"text": "Olá",
"type": "transcript.text.segment",
}
],
"usage": {"type": "duration", "seconds": 295.8},
}
result = convert_to_model_response_object(
response_object=response_object,
model_response_object=TranscriptionResponse(),
response_type="audio_transcription",
)
assert isinstance(result.usage, TranscriptionUsageDurationObject)
assert result.usage.seconds == 295.8
def test_usage_duration_object_accepts_float_seconds(self):
assert (
TranscriptionUsageDurationObject(type="duration", seconds=295.8).seconds
== 295.8
)
class TestTranscriptionDurationNotInResponseBody: