test(benchmarks): run shared logging executor inline to make CodSpeed measurements deterministic (#32435)

This commit is contained in:
Yassin Kortam 2026-07-09 11:14:22 +03:00 • committed by GitHub
parent cda99a08c8
commit 60729f733e
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 57 additions and 0 deletions

View file

@ -0,0 +1,36 @@
"""Shared setup keeping CodSpeed measurements hermetic.
CodSpeed's callgrind instrumentation counts instructions from every thread while
a measurement window is open, and valgrind serializes all threads onto one
virtual CPU. Work deferred to litellm's shared logging executor would therefore
be attributed to whichever benchmark the valgrind scheduler resumes it under,
flipping results between runs. Running the executor inline keeps each
benchmark's cost self-contained and deterministic.
"""
from collections.abc import Callable, Iterator
from concurrent.futures import Future
from typing import ParamSpec, TypeVar
import pytest
from litellm.litellm_core_utils.thread_pool_executor import executor
P = ParamSpec("P")
R = TypeVar("R")
def _submit_inline(fn: Callable[P, R], /, *args: P.args, **kwargs: P.kwargs) -> Future[R]:
future: Future[R] = Future()
try:
future.set_result(fn(*args, **kwargs))
except BaseException as exc:
future.set_exception(exc)
return future
@pytest.fixture(autouse=True, scope="session")
def inline_logging_executor() -> Iterator[None]:
executor.submit = _submit_inline
yield
del executor.submit

View file

@ -6,10 +6,13 @@ in the litellm hot path: token counting, model info lookup, provider
resolution, and cost calculation.
"""
import threading
import pytest
import litellm
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
from litellm.litellm_core_utils.thread_pool_executor import executor
from litellm.litellm_core_utils.token_counter import token_counter
@ -205,3 +208,21 @@ def test_get_model_cost_key_exact_match():
def test_get_model_cost_key_case_insensitive():
"""Benchmark model cost key lookup with case-insensitive fallback."""
litellm.utils._get_model_cost_key("GPT-4o")
# ---------------------------------------------------------------------------
# Measurement hermeticity guard
# ---------------------------------------------------------------------------
@pytest.mark.benchmark
def test_logging_executor_runs_inline():
"""Guard that the shared logging executor runs submissions inline.
Deferred submissions execute on worker threads, and callgrind attributes
their instructions to whichever benchmark's measurement window is open when
the valgrind scheduler resumes them, making results nondeterministic.
"""
future = executor.submit(threading.get_ident)
assert future.done()
assert future.result() == threading.get_ident()