feat: add CodSpeed walltime benchmarks and reliability diagnostics

This commit is contained in:
Yujong Lee 2026-09-05 17:21:44 -07:00
parent 878f0da9ab
commit 34e2ef97f0
10 changed files with 427 additions and 21 deletions

View file

@ -8,6 +8,8 @@ on:
paths:
- "litellm/**"
- "tests/benchmarks/**"
- "tests/rust-python-harness/**"
- "litellm-rust/**"
- "pyproject.toml"
- "uv.lock"
- ".github/workflows/codspeed.yml"
@ -20,6 +22,8 @@ on:
paths:
- "litellm/**"
- "tests/benchmarks/**"
- "tests/rust-python-harness/**"
- "litellm-rust/**"
- "pyproject.toml"
- "uv.lock"
- ".github/workflows/codspeed.yml"
@ -92,3 +96,60 @@ jobs:
-p pytest_codspeed.plugin
tests/benchmarks/
--codspeed
e2e-walltime:
runs-on: codspeed-macro
timeout-minutes: 30
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- uses: ./.github/actions/setup-uv-with-retries
with:
version: "0.10.9"
- uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cargo/registry
~/.cargo/git
litellm-rust/target
key: ${{ runner.os }}-${{ runner.arch }}-e2e-release-${{ hashFiles('litellm-rust/Cargo.lock') }}
restore-keys: |
${{ runner.os }}-${{ runner.arch }}-e2e-release-
- name: Build environment
run: >-
env PYTEST_DISABLE_PLUGIN_AUTOLOAD=1
uv run --frozen --no-default-groups
--with pytest==8.3.5
--with pytest-codspeed==5.0.3
--with psutil==7.2.2
--with "mcp>=1.26.0,<2.0"
--with "a2a-sdk>=1.1.0,<2.0"
pytest -p pytest_codspeed.plugin
--import-mode=importlib -o consider_namespace_packages=true
tests/rust-python-harness/strategies/e2e_benchmark/codspeed.py
--collect-only -q
- name: Build release extension
run: VIRTUAL_ENV="$PWD/.venv" uvx --from maturin==1.15.0 maturin develop --release
- name: Run OCR walltime benchmarks
uses: CodSpeedHQ/action@1c8ae4843586d3ba879736b7f6b7b0c990757fab # v4.12.1
with:
mode: walltime
run: >-
uv run --no-sync --no-default-groups
--with pytest==8.3.5
--with pytest-codspeed==5.0.3
--with psutil==7.2.2
--with "mcp>=1.26.0,<2.0"
--with "a2a-sdk>=1.1.0,<2.0"
python -m tests.rust-python-harness.strategies.e2e_benchmark.codspeed

View file

@ -58,7 +58,7 @@ tests/rust-python-harness/
- A strategy is a folder under `strategies/` with a one-line `AGENTS.md` and an `__init__.py` exporting exactly one `STRATEGY: StrategyDefinition`; its id must equal the folder name
- `shared/reporting/strategy.py` is the contract: runnable module/suite specs, not-implemented/skipped specs, the runner protocol, and `StrategyDefinition`
- Every `STRATEGY` explicitly classifies every SDK function; surface-aware strategies declare their surfaces and classify the complete surface-by-function matrix
- Run locally only; no CI integration
- Run locally; `e2e_benchmark` also has a CodSpeed walltime job
- `python -m tests.rust-python-harness run <strategy>|all` runs the selected strategy; `--function` is common, while each strategy exposes only its supported options
- Examples: `run e2e_parity --surface sdk --function ocr`, `run unit_tests_parity --function ocr --pytest-arg=-x`, or `run all --function ocr`
- `cli/catalog.py` discovers strategies, validates their Python definitions, and orders them; `cli/__init__.py` builds the Click command tree; `cli/commands.py` runs selected cases

View file

@ -1,12 +1,12 @@
# What this is
Measures Python and Rust SDK latency, process CPU time, and sampled RSS against local provider replays. Initial coverage is sync/async Mistral OCR at concurrency one. Other SDK functions remain explicitly unimplemented. Run locally, with no CI integration
Measures Python and Rust SDK latency, process CPU time, and sampled RSS against local provider replays. Initial coverage is sync/async Mistral OCR at concurrency one. Other SDK functions remain explicitly unimplemented. Keep implementation in this strategy folder. The local CLI reports latency, CPU, and RSS; the CodSpeed adapter uploads walltime measurements from the existing repository workflow
# How it works
Derive five profiles from the existing `e2e_parity` recording without editing it. `small` uses a 32 KiB PDF and one response page. `request_medium` and `request_large` increase only the PDF to 256 KiB and 2 MiB. `response_medium` and `response_large` increase only the response to 16 and 128 pages. Insert PDF comment padding before the original final `startxref` and EOF trailer, preserving object offsets. Response pages are synthetic repetitions, independent of actual PDF content
Generate fixtures in the controller, outside SDK worker imports. Each backend, route, profile, and repeat gets fresh timing and memory workers against a separate local HTTP provider process. Run Python and Rust sequentially, reversing their order on alternating repeats. The provider drains request bytes and serves preloaded responses without JSON parsing or capture. Its CPU and RSS are excluded
Generate fixtures in the controller, outside SDK worker imports. Each backend, route, profile, and repeat gets fresh timing and memory workers against a separate local HTTP provider process. Run Python and Rust sequentially, reversing their order on alternating repeats. The default four repeats balance backend order. Timing workers run at least the requested iteration count and minimum duration; the separate memory workers use the same fixed iteration count for both backends The provider drains request bytes and serves preloaded responses without JSON parsing or capture. Its CPU and RSS are excluded
Warmup, preflight response checks, bounded-memory native hashing, and garbage collection precede readiness. Require matching Python/Rust preflight digests and matching timing/memory worker digests. Every request checks the parity harness's User-Agent convention to reject Python fallback during a Rust run. Use `e2e_parity` for complete request and response semantics
@ -16,7 +16,7 @@ Sample only the SDK worker's RSS in the separate memory pass. Baseline follows w
Publish readiness/results atomically in temporary JSON files, reserving stdout/stderr for diagnostics. Bound readiness, measurement, and shutdown waits. Terminate and reap workers on failure or interruption. Reject missing extensions, backend mismatches, exceptions, and incomplete samples. Preserve completed pairs in partial reports when a worker fails
Report pooled p50/p95/p99 latency, CPU milliseconds per call, sequential calls per second, baseline/peak/after RSS, and Python p50 divided by backend p50. JSON retains raw per-repeat samples, fixture/extension hashes, Python version, options, platform, Git revision, and working-tree state. A pass confirms valid measurements without imposing performance thresholds. Use an idle host, inspect repeat variation, and treat short-run tail estimates cautiously. Loopback HTTP, allocator behavior, and deferred work affect results. Streaming, gateway overhead, live provider latency, and concurrent throughput are outside scope
Report pooled p50/p95/p99 and mean/standard-deviation latency, per-repeat sample counts, elapsed time, medians and throughput, CPU milliseconds per call, sequential calls per second, baseline/peak/after RSS, and Python p50 divided by backend p50. JSON retains raw per-repeat samples, fixture/extension hashes, Python version, options, platform, Git revision, and working-tree state. JSON schema version 2 also exports diagnostic warnings. A pass confirms completed measurements without establishing statistical significance or imposing performance thresholds. Warn for single repeats, unbalanced order, batches shorter than one second, or a repeat-median range exceeding 10% of its median. These are diagnostic heuristics; absence of warnings is not proof of stability. Pooled statistics weight each call equally, so duration-based sampling can weight faster repeats more heavily. Inspect independent repeat statistics before interpreting pooled ratios. Calls per second follows batch mean time, not median latency. Keep slow calls in the data Use an idle host, inspect repeat variation, and treat short-run tail estimates cautiously. Loopback HTTP, allocator behavior, and deferred work affect results. Streaming, gateway overhead, live provider latency, and concurrent throughput are outside scope
Editable `uv sync` builds a development extension. Build release explicitly and retain `--no-sync`:
@ -28,6 +28,23 @@ uv run --no-sync python -m tests.rust-python-harness run e2e_benchmark \
--benchmark-arg=--output=/tmp/e2e-benchmark.json
```
Forward each option through `--benchmark-arg=...`. Defaults: `--iterations=100`, `--warmup=10`, `--repeats=3`, `--timeout=120` seconds, `--sample-interval-ms=5`. Repeat `--profile=NAME` or `--route=ocr|aocr` to select subsets. `--output=PATH` exports JSON. Invalid values and unwritable destinations produce handled CLI errors. `run all` includes this strategy. A smoke run can select `small`, `ocr`, 10 iterations, 2 warmups, and 1 repeat
Forward each option through `--benchmark-arg=...`. Defaults: `--iterations=100`, `--warmup=10`, `--repeats=4`, `--min-time=1` second, `--timeout=120` seconds, `--sample-interval-ms=5`. Repeat `--profile=NAME` or `--route=ocr|aocr` to select subsets. `--output=PATH` exports JSON. Invalid values and unwritable destinations produce handled CLI errors. `run all` includes this strategy. A smoke run can select `small`, `ocr`, 10 iterations, 2 warmups, 1 repeat, and `--min-time=0`
Run focused checks with `uv run --no-sync pytest -o consider_namespace_packages=true tests/rust-python-harness/strategies/e2e_benchmark tests/rust-python-harness/cli -q`
The CodSpeed job in `.github/workflows/codspeed.yml` uses `mode: walltime` on `codspeed-macro`, with release compilation outside instrumentation. It leaves the existing CPU simulation job intact. Macro runners need explicit access to this public repository through the organization runner group. The ARM64 release cache includes runner architecture
Run the CodSpeed adapter in place:
```sh
uv pip install --python .venv/bin/python pytest-codspeed==5.0.3
uv run --no-sync python -m tests.rust-python-harness.strategies.e2e_benchmark.codspeed
```
Use `--profile=small`, `--route=ocr|aocr`, and `--max-time=5` seconds to select work. Each backend/profile/route runs once in a fresh pytest process with a stable CodSpeed benchmark ID. Setup, provider startup, preflight validation, and teardown are outside the benchmark fixture. Compare Python/Rust response digests after both processes finish. The same provider rejects backend fallback. Timed warmup and round-count selection are supplied by CodSpeed, which stores history and profiling data in CI. Repeated rounds in one process do not replace independent process repeats; use the local CLI for that audit
Set CodSpeed `min_time=0` so each round measures one full SDK call. This also avoids the pinned plugin's terminal display dividing normalized timings by iterations a second time for multi-call rounds. Async measurements await completion on a persistent loop, including `run_until_complete` entry/exit per call; local CLI async samples run inside the loop, so compare backends within each instrument rather than equating their absolute timings. Neither path overrides the logging executor. CodSpeed's best-time statistic is not the CLI's median or throughput. Memory and process CPU remain separate local measurements
Interpret results per route and profile. A synchronous small-request improvement does not establish an async improvement or monotonic scaling. Neither path isolates PyO3 overhead. Sampled peak RSS reductions do not imply equivalent retained-memory reductions. Use a separate bridge microbenchmark and allocation profiling to investigate causes
Methodology references: [Switowski](https://switowski.com/blog/how-to-benchmark-python-code/) on repeatability and setup boundaries, [CodSpeed](https://codspeed.io/docs/instruments/walltime) on I/O-inclusive measurement, [UCL](https://github-pages.arc.ucl.ac.uk/python-tooling/pages/benchmarking-profiling.html) on benchmarking versus profiling, and [Bencher](https://bencher.dev/learn/benchmarking/python/pytest-benchmark/) on distributions and mean-based throughput

View file

@ -0,0 +1,161 @@
from __future__ import annotations
import asyncio
import gc
import math
import os
import signal
import subprocess
import sys
import tempfile
from collections.abc import Awaitable, Callable
from pathlib import Path
from typing import TYPE_CHECKING, Final, cast
import click
import pytest
from litellm.llms.base_llm.ocr.transformation import OCRResponse
from .constants import PYTHON_SENTINEL
from .models import Backend, Options, Profile, Ready, Route
from .provider import provider_process
from .worker import capture_ready
from .workloads import ocr_workload
if TYPE_CHECKING:
from pytest_codspeed.plugin import BenchmarkFixture
CASES: Final = tuple(
(backend, route, profile)
for profile in Options().profiles
for route in Options().routes
for backend in ("python", "rust")
)
@pytest.mark.parametrize(("backend", "route", "profile"), CASES, ids=tuple("-".join(case) for case in CASES))
@pytest.mark.benchmark(min_time=0)
def test_sdk(benchmark: BenchmarkFixture, backend: Backend, route: Route, profile: Profile) -> None:
import litellm
assert os.environ.get("LITELLM_RUST") == ("1" if backend == "rust" else "0"), "use the codspeed module CLI"
workload: Final = ocr_workload(profile)
with provider_process(workload.response, backend) as url:
kwargs: Final = {
"model": workload.model,
"document": {"type": "document_url", "document_url": workload.document_url},
"api_key": "benchmark-local-only",
"api_base": url,
"timeout": 10,
"num_retries": 0,
}
if route == "ocr":
sync_call: Final = cast(Callable[..., OCRResponse], litellm.ocr)
ready: Final = capture_ready(sync_call(**kwargs))
def call_sync() -> None:
sync_call(**kwargs)
gc.collect()
benchmark(call_sync)
Path(os.environ["LITELLM_BENCHMARK_READY"]).write_text(ready.model_dump_json())
return
async_call: Final = cast(Callable[..., Awaitable[OCRResponse]], litellm.aocr)
loop: Final = asyncio.new_event_loop()
try:
async_ready: Final = capture_ready(loop.run_until_complete(async_call(**kwargs)))
async def call_async() -> None:
await async_call(**kwargs)
def call_on_loop() -> None:
loop.run_until_complete(call_async())
gc.collect()
benchmark(call_on_loop)
Path(os.environ["LITELLM_BENCHMARK_READY"]).write_text(async_ready.model_dump_json())
finally:
loop.run_until_complete(loop.shutdown_asyncgens())
loop.close()
def run_case(backend: Backend, route: Route, profile: Profile, ready_file: Path, max_time: float) -> Ready:
root: Final = Path(__file__).resolve().parents[4]
nodeid: Final = f"{Path(__file__).relative_to(root)}::test_sdk[{backend}-{route}-{profile}]"
with subprocess.Popen(
(
sys.executable,
"-m",
"pytest",
"-p",
"pytest_codspeed.plugin",
"-c",
os.devnull,
f"--rootdir={root}",
"--import-mode=importlib",
"-o",
"consider_namespace_packages=true",
"--codspeed",
"--codspeed-mode=walltime",
"--codspeed-warmup-time=1",
f"--codspeed-max-time={max_time}",
nodeid,
"-q",
),
cwd=root,
env={
**os.environ,
"PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1",
"LITELLM_RUST": "1" if backend == "rust" else "0",
"LITELLM_USER_AGENT": PYTHON_SENTINEL,
"LITELLM_LOCAL_MODEL_COST_MAP": "True",
"LITELLM_BENCHMARK_READY": str(ready_file),
"NO_PROXY": "127.0.0.1,localhost",
"no_proxy": "127.0.0.1,localhost",
},
start_new_session=True,
) as child:
try:
code: Final = child.wait(timeout=max_time + 120)
if code:
raise click.ClickException(f"CodSpeed benchmark failed: {backend}/{route}/{profile}, exit {code}")
except subprocess.TimeoutExpired as error:
raise click.ClickException(f"CodSpeed worker timed out: {backend}/{route}/{profile}") from error
finally:
try:
os.killpg(child.pid, signal.SIGTERM)
except ProcessLookupError:
pass
try:
child.wait(timeout=5)
except subprocess.TimeoutExpired:
os.killpg(child.pid, signal.SIGKILL)
child.wait()
return Ready.model_validate_json(ready_file.read_bytes())
def run_pair(route: Route, profile: Profile, directory: Path, max_time: float) -> None:
pair: Final = tuple(
run_case(backend, route, profile, directory / f"{backend}.json", max_time) for backend in ("python", "rust")
)
if pair[0].response_digest != pair[1].response_digest:
raise click.ClickException(f"Python/Rust responses differ for {route}/{profile}; run e2e_parity")
@click.command()
@click.option("--profile", "profiles", multiple=True, type=click.Choice(Options().profiles), default=Options().profiles)
@click.option("--route", "routes", multiple=True, type=click.Choice(Options().routes), default=Options().routes)
@click.option("--max-time", type=click.FloatRange(min=0.1, max=60), default=5.0, show_default=True)
def main(profiles: tuple[Profile, ...], routes: tuple[Route, ...], max_time: float) -> None:
"""Run isolated SDK workers with CodSpeed walltime calibration and profiling."""
if not math.isfinite(max_time):
raise click.BadParameter("must be finite", param_hint="--max-time")
with tempfile.TemporaryDirectory(prefix="litellm-codspeed-") as directory:
for profile in dict.fromkeys(profiles):
for route in dict.fromkeys(routes):
run_pair(route, profile, Path(directory), max_time)
if __name__ == "__main__":
main()

View file

@ -138,6 +138,7 @@ def benchmark(
iterations=options.iterations,
warmup=options.warmup,
phase="timing",
min_time=options.min_time,
)
ready, timing, _ = execute_phase(invocation, backend, options, repo_root)
memory_ready, _, memory = execute_phase(

View file

@ -17,7 +17,8 @@ class BenchmarkModel(BaseModel):
class Options(BenchmarkModel):
iterations: int = Field(default=100, ge=1)
warmup: int = Field(default=10, ge=1)
repeats: int = Field(default=3, ge=1)
repeats: int = Field(default=4, ge=1)
min_time: float = Field(default=1, ge=0, allow_inf_nan=False)
profiles: tuple[Profile, ...] = ("small", "request_medium", "request_large", "response_medium", "response_large")
routes: tuple[Route, ...] = ("ocr", "aocr")
timeout: float = Field(default=120, gt=0, allow_inf_nan=False)
@ -33,6 +34,7 @@ class Invocation(BenchmarkModel):
iterations: int
warmup: int
phase: Phase
min_time: float = Field(default=0, ge=0, allow_inf_nan=False)
class Ready(BenchmarkModel):

View file

@ -32,10 +32,38 @@ def measurements(results: Sequence[CaseResult]) -> tuple[Measurement, ...]:
)
def measurement_warnings(values: tuple[Measurement, ...]) -> tuple[str, ...]:
keys: Final = tuple(dict.fromkeys((value.route, value.profile, value.backend) for value in values))
def warnings(route: str, profile: str, backend: str) -> tuple[str, ...]:
group: Final = tuple(
value for value in values if (value.route, value.profile, value.backend) == (route, profile, backend)
)
medians: Final = tuple(statistics.median(value.timing.latency_ms) for value in group)
prefix: Final = f"{route}/{profile}/{backend}"
return (
*((f"{prefix}: one process repeat cannot establish repeatability",) if len(group) == 1 else ()),
*((f"{prefix}: backend order is unbalanced",) if len(group) % 2 else ()),
*(
(f"{prefix}: batch shorter than 1 second; increase --min-time",)
if any(value.timing.elapsed_ms < 1000 for value in group)
else ()
),
*(
(f"{prefix}: repeat p50 range exceeds 10% of its median; investigate variability",)
if max(medians) - min(medians) > statistics.median(medians) * 0.1
else ()
),
)
return tuple(warning for route, profile, backend in keys for warning in warnings(route, profile, backend))
def render_measurements(values: tuple[Measurement, ...]) -> str:
keys: Final = tuple(dict.fromkeys((value.route, value.profile) for value in values))
header: Final = (
"route/profile | backend | p50/p95/p99 ms | CPU ms/call | calls/s | RSS baseline/peak/after MiB | speedup"
"route/profile | backend | p50/p95/p99 ms | mean/stdev ms | CPU ms/call | calls/s | "
"RSS baseline/peak/after MiB | pooled p50 ratio"
)
def row(group: tuple[Measurement, ...], baseline: float) -> str:
@ -50,7 +78,8 @@ def render_measurements(values: tuple[Measurement, ...]) -> str:
)
return (
f"{group[0].route}/{group[0].profile} | {group[0].backend} | "
f"{median:.3f}/{percentile(samples, 0.95):.3f}/{percentile(samples, 0.99):.3f} | {cpu:.3f} | {rps:.1f} | "
f"{median:.3f}/{percentile(samples, 0.95):.3f}/{percentile(samples, 0.99):.3f} | "
f"{statistics.mean(samples):.3f}/{statistics.pstdev(samples):.3f} | {cpu:.3f} | {rps:.1f} | "
f"{'/'.join(f'{value / 2**20:.1f}' for value in rss)} | {baseline / median:.2f}x"
)
@ -64,7 +93,24 @@ def render_measurements(values: tuple[Measurement, ...]) -> str:
baseline: Final = statistics.median(sample for value in python for sample in value.timing.latency_ms)
return row(python, baseline), row(rust, baseline)
return "\n".join((header, *(line for route, profile in keys for line in rows(route, profile))))
repeats: Final = tuple(
f"{value.route}/{value.profile} | {value.backend} | {value.repeat + 1} | "
f"{len(value.timing.latency_ms)} | {value.timing.elapsed_ms:.1f} | "
f"{statistics.median(value.timing.latency_ms):.3f} | {statistics.mean(value.timing.latency_ms):.3f} | "
f"{len(value.timing.latency_ms) * 1000 / value.timing.elapsed_ms:.1f}"
for value in values
)
return "\n".join(
(
header,
*(line for route, profile in keys for line in rows(route, profile)),
"Per-repeat measurements (independent processes):",
"route/profile | backend | repeat | calls | batch ms | p50 ms | mean ms | calls/s",
*repeats,
"Completed measurements do not establish statistical significance or isolate PyO3 overhead",
*(f"WARNING: {warning}" for warning in measurement_warnings(values)),
)
)
def render_benchmark_results(results: Sequence[CaseResult]) -> tuple[ReportSection, ...]:

View file

@ -14,20 +14,21 @@ from ...shared.reporting.models import CaseResult, HarnessCase, HarnessRun, Resu
from ...shared.reporting.strategy import ModuleCaseSpec, UpdateCallback
from .execution import benchmark
from .models import Backend, BenchmarkModel, Measurement, Options, Profile, Route
from .reporting import ARTIFACT_KIND, MEASUREMENTS, measurements
from .reporting import ARTIFACT_KIND, MEASUREMENTS, measurements, measurement_warnings
if TYPE_CHECKING:
from .workloads import Workload
class Report(BenchmarkModel):
schema_version: int = 1
schema_version: int = 2
revision: str
working_tree_dirty: bool
platform: str
options: Options
measurements: tuple[Measurement, ...]
failures: tuple[tuple[str, str], ...]
warnings: tuple[str, ...] = ()
def parse_options(arguments: Sequence[str]) -> Options:
@ -36,6 +37,7 @@ def parse_options(arguments: Sequence[str]) -> Options:
"e2e_benchmark",
params=[
click.Option(("--iterations",), type=click.IntRange(min=1), default=defaults.iterations),
click.Option(("--min-time",), type=click.FloatRange(min=0), default=defaults.min_time),
click.Option(("--warmup",), type=click.IntRange(min=1), default=defaults.warmup),
click.Option(("--repeats",), type=click.IntRange(min=1), default=defaults.repeats),
click.Option(
@ -66,7 +68,7 @@ def _run_pair(
pair: Final = tuple(benchmark(workload, route, backend, repeat, options, repo_root) for backend in order)
if pair[0].ready.response_digest != pair[1].ready.response_digest:
raise ValueError("Python and Rust preflight SDK responses differ; run e2e_parity before comparing performance")
if any(len(value.timing.latency_ms) != options.iterations for value in pair):
if any(len(value.timing.latency_ms) < options.iterations for value in pair):
raise ValueError("SDK worker returned an incomplete measurement")
return pair
@ -142,6 +144,7 @@ def run_benchmark_cases(
options=options,
measurements=measurements(tuple(run.results.values())),
failures=tuple(run.failures),
warnings=measurement_warnings(measurements(tuple(run.results.values()))),
)
try:
Path(options.output).write_text(report.model_dump_json(indent=2) + "\n")

View file

@ -20,9 +20,9 @@ from litellm.llms.base_llm.ocr.transformation import OCRResponse
from ...cli import main
from .execution import execute_phase, sdk_process, wait_for_output
from .constants import PYTHON_SENTINEL
from .models import Invocation, Options, Route
from .models import Backend, Invocation, Measurement, Memory, Options, Ready, Route, Timing
from .provider import provider_process
from .reporting import percentile, render_measurements
from .reporting import measurement_warnings, percentile, render_measurements
from .runner import Report, parse_options
from .worker import file_sha256, measure_async, measure_sync
from .workloads import JSON_OBJECT, JSON_PAGES, ocr_workload, padded_pdf
@ -188,6 +188,7 @@ def test_cli_runs_both_backends_and_exports_measurements(tmp_path: Path, capsys:
"--benchmark-arg=--profile=small",
"--benchmark-arg=--route=aocr",
"--benchmark-arg=--iterations=3",
"--benchmark-arg=--min-time=0",
"--benchmark-arg=--warmup=1",
"--benchmark-arg=--repeats=2",
f"--benchmark-arg=--output={output}",
@ -203,6 +204,9 @@ def test_cli_runs_both_backends_and_exports_measurements(tmp_path: Path, capsys:
(1, "rust"),
(1, "python"),
)
assert report.schema_version == 2
assert report.warnings == measurement_warnings(report.measurements)
assert any("batch shorter than 1 second" in warning for warning in report.warnings)
assert len({value.ready.response_digest for value in report.measurements}) == 1
for value in report.measurements:
assert len(value.timing.latency_ms) == 3
@ -219,7 +223,18 @@ def test_cli_runs_both_backends_and_exports_measurements(tmp_path: Path, capsys:
@pytest.mark.parametrize(
"argument", ("--iterations=0", "--warmup=invalid", "--route=chat", "--unknown=1", "--timeout=nan", "--timeout=inf")
"argument",
(
"--iterations=0",
"--warmup=invalid",
"--route=chat",
"--unknown=1",
"--timeout=nan",
"--timeout=inf",
"--min-time=-1",
"--min-time=nan",
"--min-time=inf",
),
)
def test_cli_rejects_invalid_benchmark_options_without_a_traceback(
argument: str, capsys: pytest.CaptureFixture[str]
@ -266,3 +281,94 @@ def test_native_provenance_hash_uses_bounded_memory(tmp_path: Path) -> None:
assert peak < 1024 * 1024
finally:
tracemalloc.stop()
@pytest.mark.parametrize("route", ("ocr", "aocr"))
def test_duration_sampling_meets_time_and_iteration_minimums(route: Route) -> None:
request: Final = invocation().model_copy(update={"min_time": 0.05})
def sync_call() -> OCRResponse:
sleep(0.005)
return OCRResponse(model="benchmark", pages=[])
async def async_call() -> OCRResponse:
await asyncio.sleep(0.005)
return OCRResponse(model="benchmark", pages=[])
result: Final = (
measure_sync(sync_call, request) if route == "ocr" else asyncio.run(measure_async(async_call, request))
)
assert len(result.latency_ms) > request.iterations
assert result.elapsed_ms >= 50
count_limited: Final = request.model_copy(update={"min_time": 0.001})
short: Final = (
measure_sync(sync_call, count_limited)
if route == "ocr"
else asyncio.run(measure_async(async_call, count_limited))
)
assert len(short.latency_ms) == request.iterations
def measurement(backend: Backend, repeat: int, samples: tuple[float, ...]) -> Measurement:
return Measurement(
backend=backend,
repeat=repeat,
profile="small",
route="ocr",
document_bytes=32768,
response_bytes=100,
response_pages=1,
fixture_sha256="fixture",
ready=Ready(response_digest="response", python_version="3.12", native_sha256=None),
timing=Timing(latency_ms=samples, elapsed_ms=sum(samples), cpu_ms=1),
memory=Memory(baseline_rss_bytes=100, sampled_peak_rss_bytes=110, retained_rss_bytes=105, samples=2),
)
def test_report_exposes_tail_effects_and_insufficient_repetition() -> None:
pair: Final = (
measurement("python", 0, (1, 1, 1, 97)),
measurement("rust", 0, (1, 1, 1, 1)),
)
output: Final = render_measurements(pair)
assert "25.000/41.569" in output
assert "40.0" in output
assert "1000.0" in output
assert "Per-repeat measurements" in output
assert "one process repeat cannot establish repeatability" in output
assert "backend order is unbalanced" in output
assert "batch shorter than 1 second" in output
def test_warnings_detect_between_process_variation_without_discarding_samples() -> None:
stable: Final = tuple(measurement("python", repeat, (500, 500)) for repeat in range(4))
assert measurement_warnings(stable) == ()
varied: Final = (*stable[:3], measurement("python", 3, (1000, 1000)))
assert measurement_warnings(varied) == (
"ocr/small/python: repeat p50 range exceeds 10% of its median; investigate variability",
)
def test_codspeed_cli_runs_both_backends_and_entrypoints() -> None:
pytest.importorskip("pytest_codspeed")
result: Final = subprocess.run(
(
sys.executable,
"-m",
"tests.rust-python-harness.strategies.e2e_benchmark.codspeed",
"--profile",
"small",
"--max-time",
"0.1",
),
cwd=REPO_ROOT,
capture_output=True,
text=True,
timeout=120,
)
assert result.returncode == 0, result.stdout + result.stderr
for backend in ("python", "rust"):
for route in ("ocr", "aocr"):
assert f"test_sdk[{backend}-{route}-small]" in result.stdout
assert result.stdout.count("1 benchmarked") == 4
assert "was never awaited" not in result.stderr + result.stdout

View file

@ -6,7 +6,8 @@ import hashlib
import json
import platform
import sys
from collections.abc import Awaitable, Callable
from collections.abc import Awaitable, Callable, Iterator
from itertools import count
from pathlib import Path
from time import perf_counter_ns, process_time_ns
from typing import Final, cast
@ -24,7 +25,7 @@ def file_sha256(path: Path) -> str:
return digest.hexdigest()
def _ready(response: OCRResponse) -> Ready:
def capture_ready(response: OCRResponse) -> Ready:
from litellm.rust_bridge import get_native_bridge
from litellm.rust_bridge.configuration import rust_enabled
@ -70,6 +71,14 @@ async def _async_sample(call: Callable[[], Awaitable[OCRResponse]]) -> float:
return (perf_counter_ns() - start) / 1e6
def sample_indices(invocation: Invocation, start_ns: int) -> Iterator[int]:
deadline: Final = start_ns + invocation.min_time * 1e9
for index in count():
if index >= invocation.iterations and perf_counter_ns() >= deadline:
return
yield index
def measure_sync(call: Callable[[], OCRResponse], invocation: Invocation) -> Timing:
if invocation.phase == "memory":
for _ in range(invocation.iterations):
@ -77,7 +86,7 @@ def measure_sync(call: Callable[[], OCRResponse], invocation: Invocation) -> Tim
return Timing(latency_ms=(), cpu_ms=0, elapsed_ms=0)
cpu_start: Final = process_time_ns()
wall_start: Final = perf_counter_ns()
samples: Final = tuple(_sync_sample(call) for _ in range(invocation.iterations))
samples: Final = tuple(_sync_sample(call) for _ in sample_indices(invocation, wall_start))
elapsed: Final = perf_counter_ns() - wall_start
return Timing(latency_ms=samples, cpu_ms=(process_time_ns() - cpu_start) / 1e6, elapsed_ms=elapsed / 1e6)
@ -89,7 +98,7 @@ async def measure_async(call: Callable[[], Awaitable[OCRResponse]], invocation:
return Timing(latency_ms=(), cpu_ms=0, elapsed_ms=0)
cpu_start: Final = process_time_ns()
wall_start: Final = perf_counter_ns()
samples: Final = tuple([await _async_sample(call) for _ in range(invocation.iterations)])
samples: Final = tuple([await _async_sample(call) for _ in sample_indices(invocation, wall_start)])
elapsed: Final = perf_counter_ns() - wall_start
return Timing(latency_ms=samples, cpu_ms=(process_time_ns() - cpu_start) / 1e6, elapsed_ms=elapsed / 1e6)
@ -97,7 +106,7 @@ async def measure_async(call: Callable[[], Awaitable[OCRResponse]], invocation:
async def _run_async(call: Callable[[], Awaitable[OCRResponse]], invocation: Invocation, directory: Path) -> None:
for _ in range(invocation.warmup):
await call()
ready: Final = _ready(await call())
ready: Final = capture_ready(await call())
_handshake(ready, directory)
_finish(await measure_async(call, invocation), directory)
@ -121,7 +130,7 @@ def run_worker(invocation: Invocation, directory: Path) -> None:
call: Final = lambda: sync_route(**kwargs)
for _ in range(invocation.warmup):
call()
ready: Final = _ready(call())
ready: Final = capture_ready(call())
_handshake(ready, directory)
_finish(measure_sync(call, invocation), directory)