mirror of
https://github.com/BerriAI/litellm.git
synced 2026-08-28 05:25:59 +00:00
The floor was an absolute fleet number, so it asserted replicas x per-replica rate and went red on how many gateway pods happened to be warm rather than on the request path. The test now measures one replica first, with a short serial pass that only ever occupies a single pod, and requires the concurrent phase to reach at least that rate. A serial latency budget carries the request-path assertion the floor used to imply, and both hold at one replica or seven. Zero-error runs that "sustained 16.7 RPS" were queueing, not slow requests: the load model is a mock_response deployment with no upstream, a single-worker replica serves it in about 57ms, and 100 closed-loop users against 1/0.057 RPS of capacity sit at 6s each by Little's law. The runner also kept locust's --json summary and threw away everything else, so a run where 93% of requests failed said nothing about what they got. It now passes --csv, reads the failure breakdown back, and reports locust's own generator-saturation warnings, both folded into the assertion messages. Resolves LIT-5054
163 lines
5.8 KiB
Python
163 lines
5.8 KiB
Python
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
from locust_load import (
|
|
LoadError,
|
|
LoadResult,
|
|
LocustStatEntry,
|
|
aggregate_stats,
|
|
median_seconds,
|
|
read_errors,
|
|
read_generator_warnings,
|
|
)
|
|
|
|
_FAILURES_HEADER = "Method,Name,Error,Occurrences,First Seen,Last Seen\n"
|
|
|
|
|
|
def _entry(
|
|
*,
|
|
num_requests: int,
|
|
num_failures: int = 0,
|
|
start_time: float = 1000.0,
|
|
last_request_timestamp: float = 1010.0,
|
|
response_times: dict[int, int] | None = None,
|
|
) -> LocustStatEntry:
|
|
return LocustStatEntry(
|
|
num_requests=num_requests,
|
|
num_failures=num_failures,
|
|
start_time=start_time,
|
|
last_request_timestamp=last_request_timestamp,
|
|
response_times=response_times if response_times is not None else {50: num_requests},
|
|
)
|
|
|
|
|
|
def _result(
|
|
*,
|
|
errors: tuple[LoadError, ...] = (),
|
|
generator_warnings: tuple[str, ...] = (),
|
|
) -> LoadResult:
|
|
return LoadResult(
|
|
requests=10,
|
|
failures=10,
|
|
requests_per_second=1.0,
|
|
median_response_seconds=0.05,
|
|
errors=errors,
|
|
generator_warnings=generator_warnings,
|
|
)
|
|
|
|
|
|
class TestSerialLatency:
|
|
def test_median_is_the_middle_sample_not_the_mean_a_slow_tail_would_drag(self) -> None:
|
|
# Nine fast requests and one very slow one: the mean is 1.99s, the median is 20ms.
|
|
entry = _entry(num_requests=10, response_times={20: 9, 20000: 1})
|
|
|
|
assert median_seconds([entry]) == 0.02
|
|
|
|
def test_median_merges_the_histograms_of_every_stats_entry(self) -> None:
|
|
# Per entry the median would be 10ms and 90ms; merged, the middle of the five samples is 90ms.
|
|
entries = [
|
|
_entry(num_requests=2, response_times={10: 2}),
|
|
_entry(num_requests=3, response_times={90: 3}),
|
|
]
|
|
|
|
assert median_seconds(entries) == 0.09
|
|
|
|
def test_an_even_split_takes_the_lower_middle_sample_as_locust_itself_does(self) -> None:
|
|
entry = _entry(num_requests=4, response_times={10: 2, 90: 2})
|
|
|
|
assert median_seconds([entry]) == 0.01
|
|
|
|
def test_no_samples_reports_zero_rather_than_dividing_by_an_empty_histogram(self) -> None:
|
|
assert median_seconds([]) == 0.0
|
|
|
|
|
|
class TestAggregate:
|
|
def test_throughput_spans_the_whole_window_and_latency_comes_from_the_histogram(self) -> None:
|
|
entry = _entry(
|
|
num_requests=180,
|
|
start_time=1000.0,
|
|
last_request_timestamp=1060.0,
|
|
response_times={57: 180},
|
|
)
|
|
|
|
result = aggregate_stats([entry], (), ())
|
|
|
|
assert result.requests_per_second == 3.0
|
|
assert result.median_response_seconds == 0.057
|
|
assert result.failure_ratio == 0.0
|
|
|
|
def test_throughput_spans_from_the_earliest_start_when_locust_reports_several_entries(self) -> None:
|
|
entries = [
|
|
_entry(num_requests=60, start_time=1000.0, last_request_timestamp=1030.0),
|
|
_entry(num_requests=60, start_time=1020.0, last_request_timestamp=1060.0),
|
|
]
|
|
|
|
result = aggregate_stats(entries, (), ())
|
|
|
|
assert result.requests_per_second == 2.0
|
|
|
|
def test_a_run_that_drove_no_traffic_reports_a_total_failure_ratio(self) -> None:
|
|
result = aggregate_stats([], (), ())
|
|
|
|
assert result.requests == 0
|
|
assert result.requests_per_second == 0.0
|
|
assert result.failure_ratio == 1.0
|
|
|
|
|
|
class TestErrorBreakdown:
|
|
def test_locust_failure_rows_become_the_error_breakdown(self, tmp_path: Path) -> None:
|
|
failures_csv = tmp_path / "locust_failures.csv"
|
|
failures_csv.write_text(
|
|
_FAILURES_HEADER
|
|
+ 'POST,/chat/completions,"LocustBadStatusCode(code=401)",381,2026-07-30 12:42:01,2026-07-30 12:45:00\n'
|
|
)
|
|
|
|
assert read_errors(failures_csv) == (
|
|
LoadError(name="/chat/completions", error="LocustBadStatusCode(code=401)", occurrences=381),
|
|
)
|
|
|
|
def test_a_run_with_no_failures_writes_no_csv_and_reports_no_errors(self, tmp_path: Path) -> None:
|
|
assert read_errors(tmp_path / "locust_failures.csv") == ()
|
|
|
|
def test_diagnosis_leads_with_the_most_common_error(self) -> None:
|
|
result = _result(
|
|
errors=(
|
|
LoadError(name="/chat/completions", error="ConnectionRefused", occurrences=12),
|
|
LoadError(name="/chat/completions", error="LocustBadStatusCode(code=503)", occurrences=43675),
|
|
)
|
|
)
|
|
|
|
assert result.diagnosis().startswith("43675x /chat/completions: LocustBadStatusCode(code=503)")
|
|
|
|
def test_diagnosis_caps_the_list_and_says_how_many_it_left_out(self) -> None:
|
|
result = _result(
|
|
errors=tuple(
|
|
LoadError(name="/chat/completions", error=f"error-{index}", occurrences=index)
|
|
for index in range(1, 9)
|
|
)
|
|
)
|
|
|
|
assert result.diagnosis().count("x /chat/completions") == 5
|
|
assert "and 3 more distinct errors" in result.diagnosis()
|
|
|
|
def test_diagnosis_says_so_when_locust_recorded_nothing(self) -> None:
|
|
assert _result().diagnosis() == "locust recorded no error breakdown"
|
|
|
|
|
|
class TestGeneratorSaturation:
|
|
def test_repeated_cpu_warnings_collapse_to_one_and_reach_the_diagnosis(self) -> None:
|
|
stderr = (
|
|
"[2026-07-31 12:47:01] WARNING/locust.runners: CPU usage above 90%!\n"
|
|
"[2026-07-31 12:47:02] INFO/locust.main: Run time limit reached\n"
|
|
"[2026-07-31 12:47:03] WARNING/locust.runners: CPU usage above 90%!\n"
|
|
)
|
|
|
|
warnings = read_generator_warnings(stderr)
|
|
|
|
assert len(warnings) == 1
|
|
assert "CPU usage above 90%!" in warnings[0]
|
|
assert "CPU usage above 90%!" in _result(generator_warnings=warnings).diagnosis()
|
|
|
|
def test_ordinary_locust_chatter_is_not_reported_as_a_warning(self) -> None:
|
|
assert read_generator_warnings("[2026-07-31] INFO/locust.main: Shutting down (exit code 0)\n") == ()
|