diff --git a/litellm/integrations/otel/plumbing/providers.py b/litellm/integrations/otel/plumbing/providers.py index 8bac36aad76..8aee90ffe0a 100644 --- a/litellm/integrations/otel/plumbing/providers.py +++ b/litellm/integrations/otel/plumbing/providers.py @@ -1,5 +1,6 @@ """Provider / exporter factory + the Baggage span processor.""" +import os import queue import threading import time @@ -20,6 +21,7 @@ from opentelemetry.sdk._logs.export import ( LogExporter, SimpleLogRecordProcessor, ) +from opentelemetry.sdk.environment_variables import OTEL_METRIC_EXPORT_INTERVAL from opentelemetry.sdk.metrics import MeterProvider as SDKMeterProvider from opentelemetry.sdk.resources import Resource from opentelemetry.sdk.trace import Event, ReadableSpan, SpanProcessor, TracerProvider @@ -887,7 +889,13 @@ def build_metric_reader(config: OpenTelemetryV2Config) -> "MetricReader": ``console`` (and any unrecognized kind) exports to the console; ``otlp_http`` and ``otlp_grpc`` export over OTLP with the configured endpoint/headers. The - reader exports on a 5s period, matching v1. + reader exports on a 5s period, matching v1, unless the operator sets the + standard ``OTEL_METRIC_EXPORT_INTERVAL`` (milliseconds). Passing an explicit + interval to the SDK reader disables its own reading of that variable, so it + is resolved here: a 5s period re-ships every cumulative series the process + has ever recorded twelve times a minute, whatever the traffic, and a + per-datapoint-billed backend (Azure Monitor, Datadog) has no way to coarsen + that from its side. Histograms keep the SDK's default cumulative temporality. Prometheus-backed OTLP receivers (Grafana Cloud / Mimir, and the Prometheus OTLP endpoint) @@ -932,7 +940,39 @@ def build_metric_reader(config: OpenTelemetryV2Config) -> "MetricReader": else: exporter = ConsoleMetricExporter() - return PeriodicExportingMetricReader(exporter, export_interval_millis=5000) + return PeriodicExportingMetricReader(exporter, export_interval_millis=_metric_export_interval_millis()) + + +DEFAULT_METRIC_EXPORT_INTERVAL_MILLIS: Final = 5000 + + +def _metric_export_interval_millis() -> float: + """The metric export period: ``OTEL_METRIC_EXPORT_INTERVAL`` if set, else 5s. + + The SDK only consults the variable when no explicit interval is passed, so + the fallback to litellm's historical 5s has to live here. An unparseable + value keeps the default rather than failing the whole OTel bootstrap. + """ + raw: Final = os.environ.get(OTEL_METRIC_EXPORT_INTERVAL) + if not raw: + return DEFAULT_METRIC_EXPORT_INTERVAL_MILLIS + try: + interval: Final = float(raw) + except ValueError: + verbose_logger.warning( + "OTEL_METRIC_EXPORT_INTERVAL=%r is not a number; using %sms", + raw, + DEFAULT_METRIC_EXPORT_INTERVAL_MILLIS, + ) + return DEFAULT_METRIC_EXPORT_INTERVAL_MILLIS + if interval <= 0: + verbose_logger.warning( + "OTEL_METRIC_EXPORT_INTERVAL=%r must be positive; using %sms", + raw, + DEFAULT_METRIC_EXPORT_INTERVAL_MILLIS, + ) + return DEFAULT_METRIC_EXPORT_INTERVAL_MILLIS + return interval def _otlp_logs_endpoint(endpoint: str | None) -> str | None: diff --git a/tests/unit/integrations/otel/test_otel_v2_components.py b/tests/unit/integrations/otel/test_otel_v2_components.py index fb7be0dda14..6f6003659aa 100644 --- a/tests/unit/integrations/otel/test_otel_v2_components.py +++ b/tests/unit/integrations/otel/test_otel_v2_components.py @@ -893,6 +893,33 @@ def test_otlp_metric_exporter_uses_cumulative_histogram_temporality(): assert temporality[Histogram] is AggregationTemporality.CUMULATIVE +def test_metric_reader_default_interval_is_5s(monkeypatch): + monkeypatch.delenv("OTEL_METRIC_EXPORT_INTERVAL", raising=False) + reader = providers.build_metric_reader(OpenTelemetryV2Config(exporter="console")) + assert reader._export_interval_millis == 5000 # noqa: SLF001 # reader exposes no public accessor + + +def test_metric_reader_honours_otel_metric_export_interval(monkeypatch): + """``OTEL_METRIC_EXPORT_INTERVAL`` is the standard knob for the export period. + + The SDK reads it only when no explicit interval is passed, and litellm + passes one, so without this the variable is silently dead — and a 5s + cumulative export re-ships every series ever recorded twelve times a + minute regardless of traffic, which is what a per-datapoint-billed backend + charges for. + """ + monkeypatch.setenv("OTEL_METRIC_EXPORT_INTERVAL", "60000") + reader = providers.build_metric_reader(OpenTelemetryV2Config(exporter="console")) + assert reader._export_interval_millis == 60000 # noqa: SLF001 # reader exposes no public accessor + + +@pytest.mark.parametrize("raw", ["", "abc", "0", "-5"]) +def test_metric_reader_rejects_bad_interval(monkeypatch, raw): + monkeypatch.setenv("OTEL_METRIC_EXPORT_INTERVAL", raw) + reader = providers.build_metric_reader(OpenTelemetryV2Config(exporter="console")) + assert reader._export_interval_millis == 5000 # noqa: SLF001 # reader exposes no public accessor + + def test_otlp_logs_endpoint_normalization(): norm = providers._otlp_logs_endpoint # A base endpoint gets the signal path appended (the common OTLP env shape). @@ -1069,8 +1096,8 @@ def test_error_details_stamped_as_span_attributes_for_labels_ingest(): attributes so backends that flatten attrs into label indexes (Elastic APM ``labels.*``, Datadog span tags) render them. The exception event with the full untruncated message stays alongside.""" - from litellm.integrations.otel.model.semconv import Error, ExceptionEvent, LiteLLMError from litellm.integrations.otel.emitter import SpanEmitter + from litellm.integrations.otel.model.semconv import Error, ExceptionEvent, LiteLLMError cfg = OpenTelemetryV2Config(exporter="in_memory") provider, exporter = providers.in_memory_provider(cfg)