mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
perf(otel): resolve LITELLM_OTEL_V2 flag once instead of rebuilding settings per call (#30989)
is_otel_v2_enabled() constructed a pydantic-settings model (_OTelV2Flag) on every call, which re-scans the process environment and costs ~28us. The flag is read multiple times along the proxy request hot path (auth, logging-callback setup, proxy_server), so the cost compounded into a measurable per-request CPU overhead and a throughput regression visible from v1.87.3 onward. The flag is a process-level setting that is fixed at startup, so resolve it once with lru_cache. Caching it alone restores throughput to the pre-regression baseline in load tests. Tests that toggle the env now call cache_clear().
This commit is contained in:
parent
963816c00e
commit
1322ad7224
3 changed files with 50 additions and 0 deletions
|
|
@ -1,6 +1,7 @@
|
|||
"""Typed configuration for the OpenTelemetry instrumentation."""
|
||||
|
||||
from enum import Enum
|
||||
from functools import lru_cache
|
||||
from typing import Any, List
|
||||
|
||||
from pydantic import AliasChoices, BaseModel, Field, field_validator, model_validator
|
||||
|
|
@ -47,7 +48,12 @@ class _OTelV2Flag(BaseSettings):
|
|||
enabled: bool = Field(default=False, validation_alias=AliasChoices(OTEL_V2_ENV))
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def is_otel_v2_enabled() -> bool:
|
||||
# Resolved once at startup and cached: constructing the pydantic-settings
|
||||
# model re-scans the environment and cost ~28us, which on the proxy hot path
|
||||
# (auth, logging-callback setup) compounded into a measurable throughput
|
||||
# regression. Tests that toggle the env must call ``is_otel_v2_enabled.cache_clear()``.
|
||||
return _OTelV2Flag().enabled
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -35,6 +35,13 @@ from litellm.integrations.otel.mount import ( # noqa: E402
|
|||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clear_otel_v2_flag_cache():
|
||||
is_otel_v2_enabled.cache_clear()
|
||||
yield
|
||||
is_otel_v2_enabled.cache_clear()
|
||||
|
||||
|
||||
class _FakeSpan:
|
||||
"""Minimal recording span capturing what the hook writes."""
|
||||
|
||||
|
|
@ -70,8 +77,10 @@ def _instrumented_app():
|
|||
def test_gate_toggles_with_env(monkeypatch):
|
||||
"""The startup mount is guarded by this flag."""
|
||||
monkeypatch.delenv("LITELLM_OTEL_V2", raising=False)
|
||||
is_otel_v2_enabled.cache_clear()
|
||||
assert is_otel_v2_enabled() is False
|
||||
monkeypatch.setenv("LITELLM_OTEL_V2", "1")
|
||||
is_otel_v2_enabled.cache_clear()
|
||||
assert is_otel_v2_enabled() is True
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,8 @@
|
|||
"""Tests for the OTel v2 sources of truth: span registry, semconv keys, config,
|
||||
and the typed StandardLoggingPayload adapter. These need no OTel SDK."""
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.integrations.otel import (
|
||||
BAGGAGE_PROMOTED_KEYS,
|
||||
DB,
|
||||
|
|
@ -28,6 +30,13 @@ from litellm.integrations.otel.model.spans import (
|
|||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clear_otel_v2_flag_cache():
|
||||
is_otel_v2_enabled.cache_clear()
|
||||
yield
|
||||
is_otel_v2_enabled.cache_clear()
|
||||
|
||||
|
||||
def _sample_payload(**overrides):
|
||||
payload = {
|
||||
"call_type": "acompletion",
|
||||
|
|
@ -561,11 +570,37 @@ def test_capture_message_content_normalizer_only_touches_strings():
|
|||
|
||||
def test_v2_flag_is_off_by_default(monkeypatch):
|
||||
monkeypatch.delenv("LITELLM_OTEL_V2", raising=False)
|
||||
is_otel_v2_enabled.cache_clear()
|
||||
assert is_otel_v2_enabled() is False
|
||||
monkeypatch.setenv("LITELLM_OTEL_V2", "true")
|
||||
is_otel_v2_enabled.cache_clear()
|
||||
assert is_otel_v2_enabled() is True
|
||||
|
||||
|
||||
def test_v2_flag_resolved_once_not_per_call(monkeypatch):
|
||||
"""Regression for LIT-3895: ``is_otel_v2_enabled`` sits on the proxy hot path
|
||||
(auth, logging-callback setup). Building the pydantic-settings model on every
|
||||
call re-scanned the environment at ~28us a pop and dropped throughput, so the
|
||||
flag must be resolved once and cached rather than reconstructed per call."""
|
||||
from litellm.integrations.otel.model import config as config_mod
|
||||
|
||||
constructions = 0
|
||||
real_flag = config_mod._OTelV2Flag
|
||||
|
||||
def _counting_flag(*args, **kwargs):
|
||||
nonlocal constructions
|
||||
constructions += 1
|
||||
return real_flag(*args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(config_mod, "_OTelV2Flag", _counting_flag)
|
||||
config_mod.is_otel_v2_enabled.cache_clear()
|
||||
|
||||
for _ in range(50):
|
||||
config_mod.is_otel_v2_enabled()
|
||||
|
||||
assert constructions == 1
|
||||
|
||||
|
||||
def test_config_from_env(monkeypatch):
|
||||
for var in (
|
||||
"OTEL_EXPORTER",
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue