fix(router_strategy): serialize latency for non-chat responses in lowest-latency routing

log_success_event/async_log_success_event only converted the
end_time - start_time timedelta to float seconds inside the
isinstance(response_obj, ModelResponse) branch, so every embedding /
speech / image response appended a raw timedelta to the latency list
and broke the Redis cache sync with 'Object of type timedelta is not
JSON serializable' (no cross-replica latency sharing for those model
groups + error-log spam). Normalize response_ms to float seconds
up-front in both handlers.

Completes the partial fix from #14040. Fixes #33169

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Mihidum Hettiyahandi 2026-07-15 08:24:09 +10:00
parent 5d4c4d0fce
commit b5c363e016
2 changed files with 110 additions and 0 deletions

View file

@ -73,6 +73,13 @@ class LowestLatencyLoggingHandler(CustomLogger):
precise_minute = f"{current_date}-{current_hour}-{current_minute}"
response_ms = end_time - start_time
if isinstance(response_ms, timedelta):
# normalize to float seconds up-front: non-chat responses
# (embeddings, speech, image) skip the ModelResponse branch
# below, and a raw timedelta appended to the latency list
# breaks JSON serialization when the router cache syncs to
# Redis (issue #33169)
response_ms = response_ms.total_seconds()
time_to_first_token_response_time = None
if kwargs.get("stream", None) is not None and kwargs["stream"] is True:
@ -262,6 +269,13 @@ class LowestLatencyLoggingHandler(CustomLogger):
precise_minute = f"{current_date}-{current_hour}-{current_minute}"
response_ms = end_time - start_time
if isinstance(response_ms, timedelta):
# normalize to float seconds up-front: non-chat responses
# (embeddings, speech, image) skip the ModelResponse branch
# below, and a raw timedelta appended to the latency list
# breaks JSON serialization when the router cache syncs to
# Redis (issue #33169)
response_ms = response_ms.total_seconds()
time_to_first_token_response_time = None
if kwargs.get("stream", None) is not None and kwargs["stream"] is True:
# only log ttft for streaming request

View file

@ -0,0 +1,96 @@
#### What this tests ####
# Latency values recorded by lowest-latency routing must be JSON
# serializable for non-chat responses too (embeddings/speech/image skip
# the ModelResponse branch, so the raw timedelta used to leak into the
# latency list and break the Redis cache sync). Issue #33169.
import json
import os
import sys
from datetime import datetime, timedelta
import pytest
sys.path.insert(
0, os.path.abspath("../../..")
) # Adds the parent directory to the system path
import litellm
from litellm.caching.caching import DualCache
from litellm.router_strategy.lowest_latency import LowestLatencyLoggingHandler
DEPLOYMENT_ID = "9876"
KWARGS = {
"litellm_params": {
"metadata": {
"model_group": "gemini-embedding-001",
"deployment": "vertex_ai/gemini-embedding-001",
},
"model_info": {"id": DEPLOYMENT_ID},
}
}
def _embedding_response():
return litellm.EmbeddingResponse(
model="gemini-embedding-001",
data=[{"embedding": [0.1, 0.2], "index": 0, "object": "embedding"}],
object="list",
usage=litellm.Usage(prompt_tokens=5, completion_tokens=0, total_tokens=5),
)
def _recorded_latencies(cache: DualCache):
cached = cache.get_cache(key="gemini-embedding-001_map") or {}
return cached.get(DEPLOYMENT_ID, {}).get("latency", [])
def test_sync_embedding_latency_is_json_serializable():
"""log_success_event with datetime start/end (as the proxy passes) must not
record a raw timedelta for non-ModelResponse results."""
cache = DualCache()
handler = LowestLatencyLoggingHandler(router_cache=cache)
start_time = datetime(2026, 1, 1, 12, 0, 0)
end_time = datetime(2026, 1, 1, 12, 0, 2)
handler.log_success_event(
response_obj=_embedding_response(),
kwargs=KWARGS,
start_time=start_time,
end_time=end_time,
)
latencies = _recorded_latencies(cache)
assert latencies, "expected a latency entry to be recorded"
assert all(
not isinstance(value, timedelta) for value in latencies
), f"raw timedelta leaked into latency list: {latencies}"
assert latencies[-1] == pytest.approx(2.0)
# the exact failure mode from production: redis cache sync json.dumps
json.dumps({"latency": latencies})
@pytest.mark.asyncio
async def test_async_embedding_latency_is_json_serializable():
"""async_log_success_event is the path the proxy actually hits."""
cache = DualCache()
handler = LowestLatencyLoggingHandler(router_cache=cache)
start_time = datetime(2026, 1, 1, 12, 0, 0)
end_time = datetime(2026, 1, 1, 12, 0, 3)
await handler.async_log_success_event(
response_obj=_embedding_response(),
kwargs=KWARGS,
start_time=start_time,
end_time=end_time,
)
latencies = _recorded_latencies(cache)
assert latencies, "expected a latency entry to be recorded"
assert all(
not isinstance(value, timedelta) for value in latencies
), f"raw timedelta leaked into latency list: {latencies}"
assert latencies[-1] == pytest.approx(3.0)
json.dumps({"latency": latencies})