fix(newrelic): retry transient 429/408 metric posts instead of dropping

This commit is contained in:
Yucheng 2026-08-26 10:48:01 -07:00
parent 49f0439f54
commit 5e0d19b4a8
2 changed files with 19 additions and 1 deletions

View file

@ -61,6 +61,10 @@ from litellm.types.integrations.newrelic import (
)
from litellm.types.utils import StandardLoggingPayload
# 408 (request timeout) and 429 (rate limit) are transient client errors the
# Metric API expects a retry on, unlike 400/403 which a retry would only repeat.
_RETRYABLE_CLIENT_STATUSES: Final = frozenset({408, 429})
def resolve_newrelic_metric_endpoint(newrelic_region: str | None) -> str:
if not newrelic_region:
@ -346,7 +350,7 @@ class NewRelicMetricsLogger(CustomBatchLogger):
if 200 <= status < 300:
return True
if 400 <= status < 500:
if 400 <= status < 500 and status not in _RETRYABLE_CLIENT_STATUSES:
verbose_logger.warning(
"New Relic Metrics: %s from Metric API%s, dropping %s records.",
status,

View file

@ -770,6 +770,20 @@ async def test_raised_500_is_requeued():
assert logger.log_queue == [record], "a transient 5xx must requeue"
@pytest.mark.asyncio
@pytest.mark.parametrize("status", [429, 408])
async def test_transient_4xx_is_requeued_not_dropped(status):
"""The Metric API returns 429 when it throttles (and 408 on a request
timeout); both are transient and expect a retry, so the batch must be
requeued rather than permanently dropped like a 400/403."""
logger = _make_logger()
record = _record()
logger.log_queue.append(record)
logger.async_client.post = _raises(status)
await logger.async_send_batch()
assert logger.log_queue == [record], f"a transient {status} must requeue, not drop"
@pytest.mark.asyncio
@pytest.mark.parametrize("status", [200, 201, 204])
async def test_any_2xx_is_treated_as_delivered_not_requeued(status):