litellm/tests/integration/observability/conftest.py
mateo-berri 87b092c83d fix(prometheus): start the series cap over on a one-worker restart and audit it live
A proxy with one worker and an operator-set PROMETHEUS_MULTIPROC_DIR now drops litellm's admission files at boot, so a restart frees every slot there the way it already does with several workers. A cap or TTL that is not a number greater than 0 (a bool, a non-numeric string, an empty value) is ignored with the startup warning instead of breaking the logger

The integration cells drive the cap on every endpoint through the OpenAI and Anthropic SDKs and raw httpx, streaming and not, plus gauges, cache hits, failures, both workers of one instance, the TTL on one worker and its warning on two, ignored settings, excluded labels on the fallback counters, a null cap, /config/update, a concurrent burst scraped mid-flight, a provider outage, a killed worker, and restarts with one and two workers
2026-10-03 17:24:10 -07:00

66 lines
2.3 KiB
Python

from __future__ import annotations
import uuid
from collections.abc import Callable, Iterator, Mapping
from pathlib import Path
from typing import Final
from urllib.parse import urlparse
import pytest
import yaml
from integration._support.otlp_sink import SpanSinks, owned_sinks
from integration._support.prometheus_series import CapRig, series_cap_rig
from pydantic import JsonValue
AuditConfigWriter = Callable[[Path, Mapping[str, JsonValue]], Path]
@pytest.fixture(scope="module")
def audit_sinks(tmp_path_factory: pytest.TempPathFactory) -> Iterator[SpanSinks]:
directory: Final = tmp_path_factory.mktemp("otel-audit-sinks")
with owned_sinks(directory) as sinks:
yield sinks
@pytest.fixture(scope="module")
def otel_audit_config(audit_sinks: SpanSinks) -> AuditConfigWriter:
tenant_host: Final = urlparse(audit_sinks.tenant).netloc
def write(directory: Path, litellm_settings: Mapping[str, JsonValue] = {}) -> Path:
config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text())
config["litellm_settings"] = {
**config.get("litellm_settings", {}),
"callbacks": ["otel"],
"provider_url_destination_allowed_hosts": [tenant_host],
**dict(litellm_settings),
}
config["callback_settings"] = {
"otel": {"exporter": "http/json", "endpoint": audit_sinks.operator, "use_simple_processor": True}
}
config["general_settings"] = {**config.get("general_settings", {}), "user_api_key_cache_ttl": 2}
path: Final = directory / f"otel-audit-{uuid.uuid4().hex}.yaml"
path.write_text(yaml.safe_dump(config))
return path
return write
@pytest.fixture(scope="module")
def langfuse_vars(audit_sinks: SpanSinks) -> dict[str, JsonValue]:
return {
"langfuse_public_key": "pk-lf-audit",
"langfuse_secret_key": "sk-lf-audit",
"langfuse_host": audit_sinks.tenant,
}
@pytest.fixture(scope="session")
def capped(tmp_path_factory: pytest.TempPathFactory) -> Iterator[CapRig]:
"""A two-worker proxy capped at three series per metric, with three keys already holding a series each."""
with series_cap_rig(
tmp_path_factory.mktemp("series-cap"),
{"prometheus_metrics_max_series_per_metric": 3},
workers=2,
warm_keys=3,
) as rig:
yield rig