From 1322ad72248d53d2ebd2bcc1a1fa3a9bcf814163 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Mon, 22 Jun 2026 11:26:42 -0700 Subject: [PATCH] perf(otel): resolve LITELLM_OTEL_V2 flag once instead of rebuilding settings per call (#30989) is_otel_v2_enabled() constructed a pydantic-settings model (_OTelV2Flag) on every call, which re-scans the process environment and costs ~28us. The flag is read multiple times along the proxy request hot path (auth, logging-callback setup, proxy_server), so the cost compounded into a measurable per-request CPU overhead and a throughput regression visible from v1.87.3 onward. The flag is a process-level setting that is fixed at startup, so resolve it once with lru_cache. Caching it alone restores throughput to the pre-regression baseline in load tests. Tests that toggle the env now call cache_clear(). --- litellm/integrations/otel/model/config.py | 6 ++++ .../integrations/otel/test_otel_v2_mount.py | 9 +++++ .../otel/test_otel_v2_sources_of_truth.py | 35 +++++++++++++++++++ 3 files changed, 50 insertions(+) diff --git a/litellm/integrations/otel/model/config.py b/litellm/integrations/otel/model/config.py index a109ba898ff..991b156ae64 100644 --- a/litellm/integrations/otel/model/config.py +++ b/litellm/integrations/otel/model/config.py @@ -1,6 +1,7 @@ """Typed configuration for the OpenTelemetry instrumentation.""" from enum import Enum +from functools import lru_cache from typing import Any, List from pydantic import AliasChoices, BaseModel, Field, field_validator, model_validator @@ -47,7 +48,12 @@ class _OTelV2Flag(BaseSettings): enabled: bool = Field(default=False, validation_alias=AliasChoices(OTEL_V2_ENV)) +@lru_cache(maxsize=1) def is_otel_v2_enabled() -> bool: + # Resolved once at startup and cached: constructing the pydantic-settings + # model re-scans the environment and cost ~28us, which on the proxy hot path + # (auth, logging-callback setup) compounded into a measurable throughput + # regression. Tests that toggle the env must call ``is_otel_v2_enabled.cache_clear()``. return _OTelV2Flag().enabled diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_mount.py b/tests/test_litellm/integrations/otel/test_otel_v2_mount.py index 956d8c53cee..7240d49d022 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_mount.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_mount.py @@ -35,6 +35,13 @@ from litellm.integrations.otel.mount import ( # noqa: E402 ) +@pytest.fixture(autouse=True) +def _clear_otel_v2_flag_cache(): + is_otel_v2_enabled.cache_clear() + yield + is_otel_v2_enabled.cache_clear() + + class _FakeSpan: """Minimal recording span capturing what the hook writes.""" @@ -70,8 +77,10 @@ def _instrumented_app(): def test_gate_toggles_with_env(monkeypatch): """The startup mount is guarded by this flag.""" monkeypatch.delenv("LITELLM_OTEL_V2", raising=False) + is_otel_v2_enabled.cache_clear() assert is_otel_v2_enabled() is False monkeypatch.setenv("LITELLM_OTEL_V2", "1") + is_otel_v2_enabled.cache_clear() assert is_otel_v2_enabled() is True diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py b/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py index 3447f5bdb7e..4bb26a70b02 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py @@ -1,6 +1,8 @@ """Tests for the OTel v2 sources of truth: span registry, semconv keys, config, and the typed StandardLoggingPayload adapter. These need no OTel SDK.""" +import pytest + from litellm.integrations.otel import ( BAGGAGE_PROMOTED_KEYS, DB, @@ -28,6 +30,13 @@ from litellm.integrations.otel.model.spans import ( ) +@pytest.fixture(autouse=True) +def _clear_otel_v2_flag_cache(): + is_otel_v2_enabled.cache_clear() + yield + is_otel_v2_enabled.cache_clear() + + def _sample_payload(**overrides): payload = { "call_type": "acompletion", @@ -561,11 +570,37 @@ def test_capture_message_content_normalizer_only_touches_strings(): def test_v2_flag_is_off_by_default(monkeypatch): monkeypatch.delenv("LITELLM_OTEL_V2", raising=False) + is_otel_v2_enabled.cache_clear() assert is_otel_v2_enabled() is False monkeypatch.setenv("LITELLM_OTEL_V2", "true") + is_otel_v2_enabled.cache_clear() assert is_otel_v2_enabled() is True +def test_v2_flag_resolved_once_not_per_call(monkeypatch): + """Regression for LIT-3895: ``is_otel_v2_enabled`` sits on the proxy hot path + (auth, logging-callback setup). Building the pydantic-settings model on every + call re-scanned the environment at ~28us a pop and dropped throughput, so the + flag must be resolved once and cached rather than reconstructed per call.""" + from litellm.integrations.otel.model import config as config_mod + + constructions = 0 + real_flag = config_mod._OTelV2Flag + + def _counting_flag(*args, **kwargs): + nonlocal constructions + constructions += 1 + return real_flag(*args, **kwargs) + + monkeypatch.setattr(config_mod, "_OTelV2Flag", _counting_flag) + config_mod.is_otel_v2_enabled.cache_clear() + + for _ in range(50): + config_mod.is_otel_v2_enabled() + + assert constructions == 1 + + def test_config_from_env(monkeypatch): for var in ( "OTEL_EXPORTER",