mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
CLIENT PAUSE ALL for the length of the chaos phase instead of CLIENT PAUSE WRITE, so every Redis touchpoint on the request path times out rather than just the writes. The pause is sized to the phase because it freezes the control connection too; teardown's CLIENT UNPAUSE is a safety net for a phase that overran Latency, RSS and CPU are now budgeted as chaos-over-baseline ratios (p50/p90/p99 for latency and RSS, CPU seconds per request once) through a small phase_budget module, replacing the machine-shaped absolutes. The Redis timeout rate is reported but no longer asserted The final /metrics scrape waits for litellm_deployment_failure_responses_total to stop moving, since that counter is bumped from the async logging queue and lagged the load generator by thousands of increments. The model group carries a unique marker so a deployment left behind by an aborted run cannot absorb this run's retries Co-Authored-By: Claude Code <noreply@anthropic.com>
52 lines
2 KiB
Python
52 lines
2 KiB
Python
"""Comparing one load phase against another, for tests that degrade a dependency mid-run.
|
|
|
|
A chaos phase's absolute numbers say very little on their own: RSS scales with worker count,
|
|
latency with core count, so a ceiling calibrated on one machine is meaningless on the next.
|
|
What travels is the ratio against a healthy phase measured on the same machine in the same run.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
from typing import Final
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class Budget:
|
|
"""One metric's healthy value, its degraded value, and how much growth is allowed."""
|
|
|
|
name: str
|
|
baseline: float
|
|
degraded: float
|
|
ratio_ceiling: float
|
|
unit: str
|
|
decimals: int = 1
|
|
|
|
@property
|
|
def ratio(self) -> float | None:
|
|
"""How many times the baseline the degraded value is, or None if there is no baseline."""
|
|
return self.degraded / self.baseline if self.baseline > 0 else None
|
|
|
|
def _rendered(self, value: float) -> str:
|
|
return f"{value:.{self.decimals}f}{self.unit}"
|
|
|
|
def violation(self) -> str | None:
|
|
"""Why this metric fails its budget, or None if it passes."""
|
|
ratio: Final = self.ratio
|
|
if ratio is None:
|
|
return (
|
|
f"{self.name} measured {self._rendered(self.baseline)} in the healthy phase, so there is nothing "
|
|
f"to compare the degraded phase against; the measurement did not happen"
|
|
)
|
|
if ratio > self.ratio_ceiling:
|
|
return (
|
|
f"{self.name} went from {self._rendered(self.baseline)} healthy to "
|
|
f"{self._rendered(self.degraded)} degraded, {ratio:.1f}x the baseline and past the "
|
|
f"{self.ratio_ceiling:.1f}x allowed"
|
|
)
|
|
return None
|
|
|
|
|
|
def violations(budgets: tuple[Budget, ...]) -> tuple[str, ...]:
|
|
"""Every budget the run blew, so one failure reports all of them instead of the first."""
|
|
return tuple(violation for budget in budgets if (violation := budget.violation()) is not None)
|