mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
* fix(otel): detach post-response service spans by request phase, name redis spans by operation Service spans logged from the post-response phase (success callbacks, the response-cache write) now root their own trace linked to the request span even while the server span is still recording, instead of only when they happen to end after it. Redis service spans are named `redis <operation>`; the litellm call chain that issued them moves to the `litellm.service.caller` attribute via a typed `ServiceLoggerPayload.caller` field. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(otel): keep the service caller on failure and legacy spans, test the production phase dispatch sites Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(otel): mark anthropic messages stream cache write as post-response phase The /v1/messages streaming cache writer awaits async_add_cache inline instead of going through create_cache_write_task, so its redis span stayed parented under the request trace. Wrap the write in post_response_phase so it detaches like the chat completions write. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(anthropic): write the Messages stream cache in a background task after handoff Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
1648 lines
66 KiB
Python
1648 lines
66 KiB
Python
"""Coverage for the engine-layer components: providers/exporters, context +
|
|
baggage helpers, metrics, the typed coercion helpers, mapper branches, span-name
|
|
builders, and the registry validator's failure paths. Needs the OTel SDK."""
|
|
|
|
import contextlib
|
|
import json
|
|
import threading
|
|
import time
|
|
from collections.abc import Iterator
|
|
from contextvars import Context as ContextVarContext
|
|
from dataclasses import replace
|
|
from http.server import BaseHTTPRequestHandler, HTTPServer, ThreadingHTTPServer
|
|
|
|
import pytest
|
|
import requests
|
|
|
|
pytest.importorskip("opentelemetry")
|
|
|
|
from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ( # noqa: E402
|
|
ExportTraceServiceRequest,
|
|
)
|
|
from opentelemetry import baggage # noqa: E402
|
|
from opentelemetry.context import attach, detach # noqa: E402
|
|
from opentelemetry._logs.severity import SeverityNumber # noqa: E402
|
|
from opentelemetry.sdk._logs import LogData, LogRecord # noqa: E402
|
|
from opentelemetry.sdk._logs.export import LogExportResult # noqa: E402
|
|
from opentelemetry.sdk.metrics import MeterProvider # noqa: E402
|
|
from opentelemetry.sdk.metrics.export import InMemoryMetricReader # noqa: E402
|
|
from opentelemetry.sdk.trace import TracerProvider # noqa: E402
|
|
from opentelemetry.sdk.trace.export import ( # noqa: E402
|
|
BatchSpanProcessor,
|
|
ConsoleSpanExporter,
|
|
SimpleSpanProcessor,
|
|
)
|
|
from opentelemetry.sdk.trace.export.in_memory_span_exporter import ( # noqa: E402
|
|
InMemorySpanExporter,
|
|
)
|
|
from opentelemetry.sdk.util.instrumentation import InstrumentationScope # noqa: E402
|
|
from opentelemetry.trace import SpanKind, TraceFlags, get_current_span # noqa: E402
|
|
from opentelemetry.trace.propagation.tracecontext import ( # noqa: E402
|
|
TraceContextTextMapPropagator,
|
|
)
|
|
|
|
import litellm # noqa: E402
|
|
from tests.unit.integrations.conftest import TlsSink # noqa: E402
|
|
from litellm.integrations.otel.plumbing import context as ctx_mod # noqa: E402
|
|
from litellm.integrations.otel.plumbing import providers # noqa: E402
|
|
from litellm.integrations.otel.model.config import OpenTelemetryV2Config # noqa: E402
|
|
from litellm.integrations.otel.mappers.genai import GenAIMapper # noqa: E402
|
|
from litellm.integrations.otel.mappers.legacy import LegacyMapper # noqa: E402
|
|
from litellm.integrations.otel.plumbing.metrics import (
|
|
create_genai_metrics,
|
|
) # noqa: E402
|
|
from litellm.integrations.otel.model.payloads import ( # noqa: E402
|
|
GuardrailSpanData,
|
|
LLMCallSpanData,
|
|
LLMCost,
|
|
LLMRequestParams,
|
|
LLMUsage,
|
|
ProxyRequestSpanData,
|
|
RequestIdentity,
|
|
ServerInfo,
|
|
ServiceSpanData,
|
|
SpanError,
|
|
)
|
|
from litellm.integrations.otel.model.semconv import GenAI, GenAIOperation
|
|
from litellm.integrations.otel.model.spans import ( # noqa: E402
|
|
SPAN_REGISTRY,
|
|
LiteLLMSpanKind,
|
|
SpanRole,
|
|
SpanSpec,
|
|
db_system,
|
|
guardrail_span_name,
|
|
proxy_request_span_name,
|
|
service_span_name,
|
|
span_role_for_service,
|
|
validate_registry,
|
|
)
|
|
from litellm.integrations.otel.model.utils import ( # noqa: E402
|
|
as_bool,
|
|
as_float,
|
|
as_int,
|
|
as_str,
|
|
as_str_tuple,
|
|
)
|
|
|
|
# --- typed coercion helpers ------------------------------------------------- #
|
|
|
|
|
|
def test_as_str():
|
|
assert as_str(None) is None
|
|
assert as_str("x") == "x"
|
|
assert as_str(5) == "5"
|
|
|
|
|
|
def test_as_int():
|
|
assert as_int(True) == 1
|
|
assert as_int(3) == 3
|
|
assert as_int(3.9) == 3
|
|
assert as_int("7") == 7
|
|
assert as_int("nope") is None
|
|
assert as_int(None) is None
|
|
|
|
|
|
def test_as_float():
|
|
assert as_float(True) == 1.0
|
|
assert as_float(2) == 2.0
|
|
assert as_float("1.5") == 1.5
|
|
assert as_float("nope") is None
|
|
assert as_float(None) is None
|
|
|
|
|
|
def test_as_bool():
|
|
assert as_bool(None) is None
|
|
assert as_bool(True) is True
|
|
assert as_bool(1) is True
|
|
assert as_bool(0) is False
|
|
|
|
|
|
def test_as_str_tuple():
|
|
assert as_str_tuple(None) is None
|
|
assert as_str_tuple("a") == ("a",)
|
|
assert as_str_tuple(["a", 2]) == ("a", "2")
|
|
assert as_str_tuple(123) is None
|
|
|
|
|
|
def test_request_params_max_completion_tokens_fallback():
|
|
params = LLMRequestParams.from_model_parameters({"max_completion_tokens": 99})
|
|
assert params.max_tokens == 99
|
|
|
|
|
|
def test_server_info_from_api_base():
|
|
assert ServerInfo.from_api_base(None) is None
|
|
assert ServerInfo.from_api_base("api.host.com:8080") == ServerInfo("api.host.com", 8080)
|
|
assert ServerInfo.from_api_base("https://h.com/v1") == ServerInfo("h.com", None)
|
|
# scheme present but empty netloc -> no hostname
|
|
assert ServerInfo.from_api_base("http:///v1") is None
|
|
|
|
|
|
def test_service_span_data_from_payload():
|
|
class _Service:
|
|
value = "redis"
|
|
|
|
class _Payload:
|
|
service = _Service()
|
|
call_type = "async_set_cache"
|
|
caller = "async_set_cache <- async_add_cache"
|
|
error = None
|
|
|
|
data = ServiceSpanData.from_payload(_Payload())
|
|
assert data.service_name == "redis"
|
|
assert data.call_type == "async_set_cache"
|
|
assert data.caller == "async_set_cache <- async_add_cache"
|
|
assert data.error is None
|
|
|
|
class _FailPayload:
|
|
service = _Service()
|
|
call_type = "async_set_cache"
|
|
caller = None
|
|
error = "boom"
|
|
|
|
failed = ServiceSpanData.from_payload(_FailPayload())
|
|
assert failed.error is not None
|
|
assert failed.error.message == "boom"
|
|
|
|
|
|
# --- span name builders ----------------------------------------------------- #
|
|
|
|
|
|
def test_name_builders():
|
|
assert proxy_request_span_name(ProxyRequestSpanData("POST", "/chat/completions")) == "POST /chat/completions"
|
|
# "{service} {call_type}" so same-service calls stay distinguishable; the
|
|
# service name alone when there's no call type.
|
|
assert service_span_name(ServiceSpanData("redis", call_type="set")) == "redis set"
|
|
assert service_span_name(ServiceSpanData("redis")) == "redis"
|
|
assert guardrail_span_name(GuardrailSpanData("presidio")) == "execute_guardrail presidio"
|
|
|
|
|
|
# --- registry validator failure paths --------------------------------------- #
|
|
|
|
|
|
def test_validate_registry_detects_role_mismatch():
|
|
bad = {SpanRole.LLM_CALL: SpanSpec(SpanRole.SERVICE, LiteLLMSpanKind.CLIENT, None)}
|
|
with pytest.raises(ValueError, match="mismatched role"):
|
|
validate_registry(bad)
|
|
|
|
|
|
def test_validate_registry_detects_unknown_parent():
|
|
bad = {SpanRole.LLM_CALL: SpanSpec(SpanRole.LLM_CALL, LiteLLMSpanKind.CLIENT, parent=SpanRole.PROXY_REQUEST)}
|
|
with pytest.raises(ValueError, match="unknown parent"):
|
|
validate_registry(bad)
|
|
|
|
|
|
def test_validate_registry_detects_missing_roles():
|
|
partial = {
|
|
SpanRole.PROXY_REQUEST: SPAN_REGISTRY[SpanRole.PROXY_REQUEST],
|
|
}
|
|
with pytest.raises(ValueError, match="missing roles"):
|
|
validate_registry(partial)
|
|
|
|
|
|
# --- mappers (full branch coverage) ----------------------------------------- #
|
|
|
|
|
|
def _full_llm_call():
|
|
return LLMCallSpanData(
|
|
operation=GenAIOperation.CHAT,
|
|
provider="openai",
|
|
request_model="gpt-4o",
|
|
response_model="gpt-4o-2024",
|
|
response_id="resp_1",
|
|
request_params=LLMRequestParams(
|
|
temperature=0.7,
|
|
top_p=0.9,
|
|
top_k=40,
|
|
max_tokens=256,
|
|
frequency_penalty=0.1,
|
|
presence_penalty=0.2,
|
|
stop_sequences=("STOP",),
|
|
seed=42,
|
|
),
|
|
usage=LLMUsage(input_tokens=10, output_tokens=5, total_tokens=15),
|
|
finish_reasons=("stop",),
|
|
error=None,
|
|
response_cost=0.002,
|
|
server=ServerInfo("api.openai.com", 443),
|
|
identity=RequestIdentity(call_id="c1"),
|
|
is_streaming=True,
|
|
)
|
|
|
|
|
|
def test_genai_mapper_all_request_params():
|
|
attrs = GenAIMapper().map(_full_llm_call())
|
|
assert attrs[GenAI.REQUEST_TOP_P] == 0.9
|
|
assert attrs[GenAI.REQUEST_TOP_K] == 40
|
|
assert attrs[GenAI.REQUEST_MAX_TOKENS] == 256
|
|
assert attrs[GenAI.REQUEST_FREQUENCY_PENALTY] == 0.1
|
|
assert attrs[GenAI.REQUEST_PRESENCE_PENALTY] == 0.2
|
|
assert attrs[GenAI.REQUEST_STOP_SEQUENCES] == ["STOP"]
|
|
assert attrs[GenAI.REQUEST_SEED] == 42
|
|
assert attrs["server.port"] == 443
|
|
|
|
|
|
def test_genai_mapper_cache_token_attrs():
|
|
cached = replace(
|
|
_full_llm_call(),
|
|
usage=LLMUsage(
|
|
input_tokens=10,
|
|
output_tokens=5,
|
|
total_tokens=15,
|
|
cache_creation_input_tokens=7,
|
|
cache_read_input_tokens=3,
|
|
),
|
|
)
|
|
attrs = GenAIMapper().map(cached)
|
|
assert attrs[GenAI.USAGE_CACHE_CREATION_INPUT_TOKENS] == 7
|
|
assert attrs[GenAI.USAGE_CACHE_READ_INPUT_TOKENS] == 3
|
|
|
|
# No cache usage keeps the span sparse: neither key present.
|
|
uncached = GenAIMapper().map(_full_llm_call())
|
|
assert GenAI.USAGE_CACHE_CREATION_INPUT_TOKENS not in uncached
|
|
assert GenAI.USAGE_CACHE_READ_INPUT_TOKENS not in uncached
|
|
|
|
|
|
def test_genai_mapper_stamps_input_output_messages():
|
|
data = LLMCallSpanData(
|
|
operation=GenAIOperation.CHAT,
|
|
provider="openai",
|
|
request_model="gpt-4o",
|
|
response_model="gpt-4o-2024",
|
|
response_id="resp_1",
|
|
request_params=LLMRequestParams(),
|
|
usage=LLMUsage(),
|
|
finish_reasons=("stop",),
|
|
error=None,
|
|
response_cost=None,
|
|
server=None,
|
|
identity=RequestIdentity(call_id="c1"),
|
|
messages_in=(
|
|
{"role": "system", "content": "Be concise."},
|
|
{"role": "user", "content": "What's the weather?"},
|
|
),
|
|
choices_out=(
|
|
{
|
|
"finish_reason": "stop",
|
|
"message": {"role": "assistant", "content": "Sunny."},
|
|
},
|
|
),
|
|
)
|
|
attrs = GenAIMapper().map(data)
|
|
assert json.loads(attrs[GenAI.INPUT_MESSAGES]) == [
|
|
{"role": "system", "content": "Be concise."},
|
|
{"role": "user", "content": "What's the weather?"},
|
|
]
|
|
assert json.loads(attrs[GenAI.OUTPUT_MESSAGES]) == [{"role": "assistant", "content": "Sunny."}]
|
|
|
|
|
|
def test_genai_mapper_omits_messages_when_content_not_captured():
|
|
attrs = GenAIMapper().map(_full_llm_call())
|
|
assert GenAI.INPUT_MESSAGES not in attrs
|
|
assert GenAI.OUTPUT_MESSAGES not in attrs
|
|
|
|
|
|
def test_genai_mapper_cost_breakdown():
|
|
from litellm.integrations.otel.model.semconv import LiteLLM
|
|
|
|
data = LLMCallSpanData(
|
|
operation=GenAIOperation.CHAT,
|
|
provider="anthropic",
|
|
request_model="claude-sonnet-4-6",
|
|
response_model=None,
|
|
response_id=None,
|
|
request_params=LLMRequestParams(),
|
|
usage=LLMUsage(),
|
|
finish_reasons=(),
|
|
error=None,
|
|
response_cost=0.012,
|
|
server=None,
|
|
identity=RequestIdentity(call_id=None),
|
|
cost=LLMCost(
|
|
input=0.004,
|
|
output=0.006,
|
|
cache_read=0.001,
|
|
cache_creation=0.0,
|
|
tool_usage=0.0005,
|
|
original=0.013,
|
|
discount_amount=0.001,
|
|
discount_percent=0.077,
|
|
margin_total_amount=0.0,
|
|
# margin_fixed_amount / margin_percent left unset on purpose
|
|
),
|
|
)
|
|
attrs = GenAIMapper().map(data)
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}total"] == 0.012
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}input"] == 0.004
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}output"] == 0.006
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}cache_read"] == 0.001
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}cache_creation"] == 0.0
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}tool_usage"] == 0.0005
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}original"] == 0.013
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}discount_amount"] == 0.001
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}discount_percent"] == 0.077
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}margin_total_amount"] == 0.0
|
|
# Components the source did not report are omitted, not zero-filled.
|
|
assert f"{LiteLLM.COST_PREFIX}margin_fixed_amount" not in attrs
|
|
assert f"{LiteLLM.COST_PREFIX}margin_percent" not in attrs
|
|
|
|
|
|
def test_genai_mapper_cost_breakdown_absent():
|
|
# No cost_breakdown → only the rolled-up total (from response_cost) emits.
|
|
from litellm.integrations.otel.model.semconv import LiteLLM
|
|
|
|
attrs = GenAIMapper().map(_full_llm_call())
|
|
assert attrs[f"{LiteLLM.COST_PREFIX}total"] == 0.002
|
|
assert not any(k.startswith(LiteLLM.COST_PREFIX) and k != f"{LiteLLM.COST_PREFIX}total" for k in attrs)
|
|
|
|
|
|
def test_llm_cost_from_breakdown_maps_costbreakdown_keys():
|
|
cost = LLMCost.from_breakdown(
|
|
{
|
|
"input_cost": 0.004,
|
|
"output_cost": 0.006,
|
|
"cache_read_cost": 0.001,
|
|
"cache_creation_cost": 0.002,
|
|
"tool_usage_cost": 0.0005,
|
|
"original_cost": 0.013,
|
|
"discount_amount": 0.001,
|
|
"discount_percent": 0.077,
|
|
"margin_fixed_amount": 0.0,
|
|
"margin_percent": 0.1,
|
|
"margin_total_amount": 0.0011,
|
|
"total_cost": 0.012, # carried on response_cost, not LLMCost
|
|
}
|
|
)
|
|
assert cost.input == 0.004
|
|
assert cost.output == 0.006
|
|
assert cost.cache_read == 0.001
|
|
assert cost.cache_creation == 0.002
|
|
assert cost.tool_usage == 0.0005
|
|
assert cost.original == 0.013
|
|
assert cost.discount_amount == 0.001
|
|
assert cost.discount_percent == 0.077
|
|
assert cost.margin_fixed_amount == 0.0
|
|
assert cost.margin_percent == 0.1
|
|
assert cost.margin_total_amount == 0.0011
|
|
|
|
|
|
def test_llm_cost_from_breakdown_none_is_empty():
|
|
assert LLMCost.from_breakdown(None) == LLMCost()
|
|
|
|
|
|
def test_genai_mapper_guardrail_and_service():
|
|
from litellm.integrations.otel.model.semconv import LiteLLM
|
|
|
|
g = GenAIMapper().map(GuardrailSpanData("presidio", mode="pre"))
|
|
assert g[LiteLLM.GUARDRAIL_NAME] == "presidio"
|
|
assert g[LiteLLM.GUARDRAIL_MODE] == "pre"
|
|
|
|
# A datastore service (redis) also gets db.* semconv.
|
|
s = GenAIMapper().map(ServiceSpanData("redis", call_type="set"))
|
|
assert s[LiteLLM.SERVICE_NAME] == "redis"
|
|
assert s[LiteLLM.SERVICE_CALL_TYPE] == "set"
|
|
assert s["db.system.name"] == "redis"
|
|
assert s["db.operation.name"] == "set"
|
|
|
|
# An internal service (router) gets no db.* keys.
|
|
internal = GenAIMapper().map(ServiceSpanData("router", call_type="acompletion"))
|
|
assert internal[LiteLLM.SERVICE_NAME] == "router"
|
|
assert "db.system.name" not in internal
|
|
|
|
|
|
def test_genai_mapper_guardrail_billing_attrs():
|
|
"""Billing counters and USD cost stamped on StandardLoggingGuardrailInformation
|
|
surface on the guardrail span: usage JSON-serialized, cost numeric under the
|
|
litellm.cost.* namespace."""
|
|
from litellm.integrations.otel.model.semconv import LiteLLM
|
|
|
|
entry = {
|
|
"guardrail_name": "azure-shield",
|
|
"guardrail_status": "success",
|
|
"guardrail_usage": {"requests": 2, "input_characters": 12000, "text_records": 12},
|
|
"guardrail_cost": 0.00456,
|
|
}
|
|
data = GuardrailSpanData.from_logging_entry(entry)
|
|
assert data.cost == 0.00456
|
|
assert data.usage_json is not None and '"text_records": 12' in data.usage_json
|
|
|
|
attrs = GenAIMapper().map(data)
|
|
assert attrs[LiteLLM.GUARDRAIL_COST] == 0.00456
|
|
assert LiteLLM.GUARDRAIL_COST == "litellm.cost.guardrail"
|
|
assert attrs[LiteLLM.GUARDRAIL_USAGE] == data.usage_json
|
|
|
|
# A guardrail without billing data keeps a sparse span: neither key present.
|
|
unbilled = GenAIMapper().map(GuardrailSpanData("presidio", mode="pre"))
|
|
assert LiteLLM.GUARDRAIL_COST not in unbilled
|
|
assert LiteLLM.GUARDRAIL_USAGE not in unbilled
|
|
|
|
|
|
def test_legacy_mapper_all_request_params():
|
|
attrs = LegacyMapper().map(_full_llm_call())
|
|
assert attrs["llm.top_k"] == 40
|
|
assert attrs["llm.frequency_penalty"] == 0.1
|
|
assert attrs["llm.presence_penalty"] == 0.2
|
|
assert attrs["llm.chat.stop_sequences"] == ["STOP"]
|
|
assert attrs["gen_ai.usage.total_tokens"] == 15
|
|
|
|
|
|
def test_legacy_mapper_covers_service_with_v1_bare_keys():
|
|
"""Service spans dual-emit V1's bare ``service``/``call_type``/``error`` keys."""
|
|
attrs = LegacyMapper().map(
|
|
ServiceSpanData("redis", call_type="set", caller="set <- add", event_metadata={"k": "v"}),
|
|
)
|
|
assert attrs["service"] == "redis"
|
|
assert attrs["call_type"] == "set"
|
|
assert attrs["caller"] == "set <- add"
|
|
assert attrs["k"] == "v" # event_metadata is stamped bare (V1 behavior)
|
|
|
|
|
|
def test_legacy_mapper_skips_guardrail_role():
|
|
"""Guardrail spans never had a V1 vocabulary; legacy mapper returns ``{}``."""
|
|
assert LegacyMapper().map(GuardrailSpanData("presidio")) == {}
|
|
|
|
|
|
# --- metrics ---------------------------------------------------------------- #
|
|
|
|
|
|
def test_create_genai_metrics_records():
|
|
reader = InMemoryMetricReader()
|
|
meter = MeterProvider(metric_readers=[reader]).get_meter("test")
|
|
metrics = create_genai_metrics(meter)
|
|
metrics.token_usage.record(10, {"x": "y"})
|
|
metrics.operation_duration.record(0.5, {"x": "y"})
|
|
data = reader.get_metrics_data()
|
|
assert data is not None
|
|
|
|
|
|
# --- context + baggage helpers ---------------------------------------------- #
|
|
|
|
|
|
def test_extract_traceparent():
|
|
valid = {"traceparent": "00-0af7651916cd43dd8448eb211c80319c-b7ad6b7169203331-01"}
|
|
assert ctx_mod.extract_traceparent(valid) is not None
|
|
assert ctx_mod.extract_traceparent({"x": "y"}) is None
|
|
|
|
|
|
def _test_tracer():
|
|
exporter = InMemorySpanExporter()
|
|
provider = TracerProvider()
|
|
provider.add_span_processor(SimpleSpanProcessor(exporter))
|
|
return provider.get_tracer("test")
|
|
|
|
|
|
_CALLER_TRACEPARENT = "00-11111111111111111111111111111111-2222222222222222-01"
|
|
|
|
|
|
def test_inject_trace_context_prefers_request_root_span():
|
|
def run():
|
|
tracer = _test_tracer()
|
|
inbound = TraceContextTextMapPropagator().extract({"traceparent": _CALLER_TRACEPARENT})
|
|
with tracer.start_as_current_span("root", context=inbound) as root:
|
|
ctx_mod.set_request_root_span(root)
|
|
result = ctx_mod.inject_trace_context({"traceparent": _CALLER_TRACEPARENT})
|
|
propagated = get_current_span(TraceContextTextMapPropagator().extract(result))
|
|
return result, root, propagated
|
|
|
|
result, root, propagated = ContextVarContext().run(run)
|
|
assert result["traceparent"] != _CALLER_TRACEPARENT
|
|
assert propagated.get_span_context().trace_id == root.get_span_context().trace_id
|
|
assert propagated.get_span_context().span_id == root.get_span_context().span_id
|
|
|
|
|
|
def test_inject_trace_context_uses_ambient_span_without_request_root():
|
|
def run():
|
|
tracer = _test_tracer()
|
|
with tracer.start_as_current_span("ambient") as ambient:
|
|
result = ctx_mod.inject_trace_context({})
|
|
propagated = get_current_span(TraceContextTextMapPropagator().extract(result))
|
|
return ambient, propagated
|
|
|
|
ambient, propagated = ContextVarContext().run(run)
|
|
assert propagated.get_span_context().trace_id == ambient.get_span_context().trace_id
|
|
assert propagated.get_span_context().span_id == ambient.get_span_context().span_id
|
|
|
|
|
|
def test_inject_trace_context_replaces_same_trace_headers_with_request_span():
|
|
def run():
|
|
tracer = _test_tracer()
|
|
headers = {"Traceparent": _CALLER_TRACEPARENT, "Tracestate": "vendor=caller", "x-keep": "1"}
|
|
inbound = TraceContextTextMapPropagator().extract({key.lower(): value for key, value in headers.items()})
|
|
with tracer.start_as_current_span("ambient", context=inbound) as ambient:
|
|
result = ctx_mod.inject_trace_context(headers)
|
|
propagated = get_current_span(TraceContextTextMapPropagator().extract(result))
|
|
return result, ambient, propagated
|
|
|
|
result, ambient, propagated = ContextVarContext().run(run)
|
|
assert sum(key.lower() == "traceparent" for key in result) == 1
|
|
assert sum(key.lower() == "tracestate" for key in result) == 1
|
|
assert result["x-keep"] == "1"
|
|
assert result["tracestate"] == "vendor=caller"
|
|
assert propagated.get_span_context().span_id == ambient.get_span_context().span_id
|
|
|
|
|
|
def test_inject_trace_context_keeps_caller_traceparent_from_another_trace():
|
|
def run():
|
|
tracer = _test_tracer()
|
|
parent = tracer.start_span("litellm_request")
|
|
with tracer.start_as_current_span("ambient") as ambient:
|
|
ctx_mod.set_request_root_span(ambient)
|
|
headers = {"Traceparent": _CALLER_TRACEPARENT, "Tracestate": "vendor=caller", "x-keep": "1"}
|
|
result = ctx_mod.inject_trace_context(headers, parent_span=parent)
|
|
propagated = get_current_span(TraceContextTextMapPropagator().extract(result))
|
|
return result, parent, propagated
|
|
|
|
result, parent, propagated = ContextVarContext().run(run)
|
|
assert result["traceparent"] == _CALLER_TRACEPARENT
|
|
assert result["tracestate"] == "vendor=caller"
|
|
assert result["x-keep"] == "1"
|
|
assert sum(key.lower() == "traceparent" for key in result) == 1
|
|
assert sum(key.lower() == "tracestate" for key in result) == 1
|
|
assert propagated.get_span_context().trace_id != parent.get_span_context().trace_id
|
|
|
|
|
|
def test_inject_trace_context_replaces_malformed_caller_traceparent():
|
|
def run():
|
|
tracer = _test_tracer()
|
|
parent = tracer.start_span("litellm_request")
|
|
with tracer.start_as_current_span("ambient"):
|
|
headers = {"traceparent": "not-a-traceparent", "tracestate": "vendor=caller"}
|
|
result = ctx_mod.inject_trace_context(headers, parent_span=parent)
|
|
propagated = get_current_span(TraceContextTextMapPropagator().extract(result))
|
|
return result, parent, propagated
|
|
|
|
result, parent, propagated = ContextVarContext().run(run)
|
|
assert propagated.get_span_context().span_id == parent.get_span_context().span_id
|
|
assert "tracestate" not in result
|
|
|
|
|
|
def test_inject_trace_context_prefers_explicit_parent_span_over_root_and_ambient():
|
|
def run():
|
|
tracer = _test_tracer()
|
|
parent = tracer.start_span("litellm_request")
|
|
with tracer.start_as_current_span("ambient") as ambient:
|
|
ctx_mod.set_request_root_span(ambient)
|
|
result = ctx_mod.inject_trace_context({}, parent_span=parent)
|
|
propagated = get_current_span(TraceContextTextMapPropagator().extract(result))
|
|
return parent, ambient, propagated
|
|
|
|
parent, ambient, propagated = ContextVarContext().run(run)
|
|
assert propagated.get_span_context().trace_id == parent.get_span_context().trace_id
|
|
assert propagated.get_span_context().span_id == parent.get_span_context().span_id
|
|
assert propagated.get_span_context().span_id != ambient.get_span_context().span_id
|
|
|
|
|
|
def test_inject_trace_context_skips_unusable_parent_span():
|
|
def run():
|
|
tracer = _test_tracer()
|
|
with tracer.start_as_current_span("ambient") as ambient:
|
|
result = ctx_mod.inject_trace_context({}, parent_span=object())
|
|
propagated = get_current_span(TraceContextTextMapPropagator().extract(result))
|
|
return ambient, propagated
|
|
|
|
ambient, propagated = ContextVarContext().run(run)
|
|
assert propagated.get_span_context().span_id == ambient.get_span_context().span_id
|
|
|
|
|
|
def test_inject_trace_context_returns_headers_unchanged_without_context():
|
|
headers = {"x-custom": "value"}
|
|
|
|
result = ContextVarContext().run(lambda: ctx_mod.inject_trace_context(headers))
|
|
|
|
assert result == headers
|
|
assert "traceparent" not in result
|
|
assert result is not headers
|
|
|
|
|
|
def test_inject_trace_context_does_not_forward_baggage():
|
|
def run():
|
|
tracer = _test_tracer()
|
|
with tracer.start_as_current_span("ambient"):
|
|
token = attach(baggage.set_baggage("litellm.team.id", "team"))
|
|
try:
|
|
return ctx_mod.inject_trace_context({})
|
|
finally:
|
|
detach(token)
|
|
|
|
result = ContextVarContext().run(run)
|
|
assert "baggage" not in result
|
|
|
|
|
|
def test_set_request_baggage_empty_returns_context():
|
|
assert ctx_mod.set_request_baggage({}) is not None
|
|
|
|
|
|
def test_get_baggage_attributes_roundtrip():
|
|
ctx = ctx_mod.set_request_baggage({"litellm.team.id": "t1"})
|
|
assert ctx_mod.get_baggage_attributes(ctx)["litellm.team.id"] == "t1"
|
|
|
|
|
|
# --- providers -------------------------------------------------------------- #
|
|
|
|
|
|
def test_to_otel_span_kind_covers_all():
|
|
assert providers.to_otel_span_kind(LiteLLMSpanKind.SERVER) is SpanKind.SERVER
|
|
assert providers.to_otel_span_kind(LiteLLMSpanKind.CLIENT) is SpanKind.CLIENT
|
|
assert providers.to_otel_span_kind(LiteLLMSpanKind.INTERNAL) is SpanKind.INTERNAL
|
|
assert providers.to_otel_span_kind(LiteLLMSpanKind.PRODUCER) is SpanKind.PRODUCER
|
|
assert providers.to_otel_span_kind(LiteLLMSpanKind.CONSUMER) is SpanKind.CONSUMER
|
|
|
|
|
|
def test_parse_headers():
|
|
assert providers.parse_headers(None) == {}
|
|
assert providers.parse_headers("a=1,b=2") == {"a": "1", "b": "2"}
|
|
assert providers.parse_headers("no-equals") == {}
|
|
|
|
|
|
def test_parse_headers_percent_decodes_values():
|
|
"""A percent-encoded OTLP header value reaches the exporter decoded.
|
|
|
|
``OTEL_EXPORTER_OTLP_HEADERS`` is W3C Baggage encoded, and Grafana Cloud
|
|
documents ``Authorization=Basic%20<token>``. Forwarding the literal ``%20``
|
|
makes the backend reject the export as a malformed credential.
|
|
"""
|
|
token = "MTMzNzc4MzpnbGNfZXlKdklqb2lNVEl6TkNJPQ=="
|
|
assert providers.parse_headers(f"Authorization=Basic%20{token}") == {"authorization": f"Basic {token}"}
|
|
assert providers.parse_headers("x-scope-orgid=team%20a") == {"x-scope-orgid": "team a"}
|
|
|
|
|
|
def test_parse_headers_keeps_unencoded_values_working():
|
|
"""Values that are not percent-encoded keep parsing unchanged.
|
|
|
|
Vendors that document a bare space, and litellm's own presets, must survive
|
|
the switch to the spec-compliant parser. Base64 padding also means a value
|
|
can contain ``=``, so only the first one may split the pair.
|
|
"""
|
|
assert providers.parse_headers("Authorization=Bearer sk-123") == {"authorization": "Bearer sk-123"}
|
|
assert providers.parse_headers("api_key=abc,space_id=xyz") == {"api_key": "abc", "space_id": "xyz"}
|
|
assert providers.parse_headers("api_key=YWJjZA==") == {"api_key": "YWJjZA=="}
|
|
|
|
|
|
def test_otlp_traces_endpoint_normalization():
|
|
norm = providers._otlp_traces_endpoint
|
|
# A base endpoint gets the signal path appended (the common OTLP env shape).
|
|
assert norm("http://collector:4318") == "http://collector:4318/v1/traces"
|
|
assert norm("http://collector:4318/") == "http://collector:4318/v1/traces"
|
|
# An already-correct path is left intact.
|
|
assert norm("http://collector:4318/v1/traces") == "http://collector:4318/v1/traces"
|
|
# Another signal's path is rewritten to traces.
|
|
assert norm("http://collector:4318/v1/logs") == "http://collector:4318/v1/traces"
|
|
# Splunk's path is preserved; None passes through.
|
|
assert norm("https://x.splunk.com/v2/trace/otlp") == "https://x.splunk.com/v2/trace/otlp"
|
|
assert norm(None) is None
|
|
|
|
|
|
def test_build_span_exporter_variants():
|
|
assert isinstance(
|
|
providers.build_span_exporter(OpenTelemetryV2Config(exporter="console")),
|
|
ConsoleSpanExporter,
|
|
)
|
|
assert isinstance(
|
|
providers.build_span_exporter(OpenTelemetryV2Config(exporter="in_memory")),
|
|
InMemorySpanExporter,
|
|
)
|
|
assert isinstance(
|
|
providers.build_span_exporter(OpenTelemetryV2Config(exporter="unknown")),
|
|
ConsoleSpanExporter,
|
|
)
|
|
http_exporter = providers.build_span_exporter(OpenTelemetryV2Config(exporter="otlp_http", endpoint="http://h:4318"))
|
|
assert "OTLPSpanExporter" in type(http_exporter).__name__
|
|
|
|
|
|
def _export_one_trace_to_local_collector(exporter_kind: str) -> tuple[list[dict], tuple[int, int, int]]:
|
|
"""Run a parent/child trace through the configured exporter against a
|
|
throwaway HTTP collector. Returns the requests as the collector saw them
|
|
(child first, since it ends first) and (trace_id, parent span_id, child span_id)."""
|
|
received: list[dict] = []
|
|
|
|
class Collector(BaseHTTPRequestHandler):
|
|
def do_POST(self):
|
|
body = self.rfile.read(int(self.headers["Content-Length"]))
|
|
received.append({"path": self.path, "headers": dict(self.headers), "body": body})
|
|
self.send_response(200)
|
|
self.end_headers()
|
|
|
|
def log_message(self, *_args):
|
|
pass
|
|
|
|
server = HTTPServer(("127.0.0.1", 0), Collector)
|
|
threading.Thread(target=server.serve_forever, daemon=True).start()
|
|
try:
|
|
config = OpenTelemetryV2Config(
|
|
exporter=exporter_kind,
|
|
endpoint=f"http://127.0.0.1:{server.server_port}",
|
|
headers="x-collector-token=secret",
|
|
)
|
|
provider = TracerProvider()
|
|
provider.add_span_processor(SimpleSpanProcessor(providers.build_span_exporter(config)))
|
|
tracer = provider.get_tracer("test")
|
|
with tracer.start_as_current_span("parent", kind=SpanKind.SERVER) as parent:
|
|
with tracer.start_as_current_span("child") as child:
|
|
ids = (
|
|
parent.get_span_context().trace_id,
|
|
parent.get_span_context().span_id,
|
|
child.get_span_context().span_id,
|
|
)
|
|
provider.shutdown()
|
|
finally:
|
|
server.shutdown()
|
|
server.server_close()
|
|
assert len(received) == 2
|
|
return received, ids
|
|
|
|
|
|
def _only_span(request: dict) -> dict:
|
|
scope_spans = json.loads(request["body"])["resourceSpans"][0]["scopeSpans"][0]["spans"]
|
|
assert len(scope_spans) == 1
|
|
return scope_spans[0]
|
|
|
|
|
|
def test_http_json_exporter_posts_otlp_json_to_traces_endpoint():
|
|
"""``http/json`` must put the OTLP/JSON mapping on the wire (camelCase
|
|
fields, integer enums, hex ids) with a JSON content type, so collectors that
|
|
cannot decode protobuf can ingest the trace. Headers still travel."""
|
|
(child_request, parent_request), (trace_id, parent_id, child_id) = _export_one_trace_to_local_collector("http/json")
|
|
|
|
assert parent_request["path"] == "/v1/traces"
|
|
assert parent_request["headers"]["Content-Type"] == "application/json"
|
|
assert parent_request["headers"]["x-collector-token"] == "secret"
|
|
parent = _only_span(parent_request)
|
|
assert parent["name"] == "parent"
|
|
assert parent["kind"] == 2
|
|
assert parent["traceId"] == format(trace_id, "032x")
|
|
assert parent["spanId"] == format(parent_id, "016x")
|
|
assert "parentSpanId" not in parent
|
|
child = _only_span(child_request)
|
|
assert child["traceId"] == format(trace_id, "032x")
|
|
assert child["spanId"] == format(child_id, "016x")
|
|
assert child["parentSpanId"] == format(parent_id, "016x")
|
|
|
|
|
|
def test_http_protobuf_exporter_still_posts_protobuf():
|
|
(_child_request, parent_request), (trace_id, _parent_id, _child_id) = _export_one_trace_to_local_collector(
|
|
"http/protobuf"
|
|
)
|
|
|
|
assert parent_request["path"] == "/v1/traces"
|
|
assert parent_request["headers"]["Content-Type"] == "application/x-protobuf"
|
|
assert format(trace_id, "032x").encode() not in parent_request["body"]
|
|
decoded = ExportTraceServiceRequest.FromString(parent_request["body"])
|
|
span = decoded.resource_spans[0].scope_spans[0].spans[0]
|
|
assert span.name == "parent"
|
|
assert span.trace_id == trace_id.to_bytes(16, "big")
|
|
|
|
|
|
@pytest.fixture
|
|
def otlp_collector() -> Iterator[tuple[str, list[str]]]:
|
|
received_paths: list[str] = []
|
|
|
|
class RecordingHandler(BaseHTTPRequestHandler):
|
|
def do_POST(self) -> None:
|
|
self.rfile.read(int(self.headers.get("Content-Length", "0")))
|
|
received_paths.append(self.path)
|
|
self.send_response(200)
|
|
self.end_headers()
|
|
|
|
def log_message(self, format: str, *args: object) -> None:
|
|
pass
|
|
|
|
server = ThreadingHTTPServer(("127.0.0.1", 0), RecordingHandler)
|
|
threading.Thread(target=server.serve_forever, daemon=True).start()
|
|
try:
|
|
yield f"http://127.0.0.1:{server.server_port}", received_paths
|
|
finally:
|
|
server.shutdown()
|
|
server.server_close()
|
|
|
|
|
|
def _export_one_span(cfg: OpenTelemetryV2Config) -> None:
|
|
provider = providers.build_tracer_provider(cfg)
|
|
provider.get_tracer("probe").start_span("probe").end()
|
|
assert provider.force_flush()
|
|
provider.shutdown()
|
|
|
|
|
|
def test_traces_endpoint_env_posts_to_the_configured_url_verbatim(monkeypatch, otlp_collector):
|
|
base_url, received_paths = otlp_collector
|
|
for var in ("OTEL_EXPORTER", "OTEL_EXPORTER_OTLP_PROTOCOL", "OTEL_EXPORTER_OTLP_ENDPOINT"):
|
|
monkeypatch.delenv(var, raising=False)
|
|
monkeypatch.setenv("OTEL_ENDPOINT", f"{base_url}/services/collector")
|
|
monkeypatch.setenv("OTEL_TRACES_ENDPOINT", f"{base_url}/services/collector/traces")
|
|
|
|
cfg = OpenTelemetryV2Config.from_env()
|
|
assert cfg.exporter == "otlp_http"
|
|
_export_one_span(cfg)
|
|
assert received_paths == ["/services/collector/traces"]
|
|
|
|
|
|
def test_traces_endpoint_alias_alone_implies_otlp_http(monkeypatch, otlp_collector):
|
|
base_url, received_paths = otlp_collector
|
|
for var in ("OTEL_EXPORTER", "OTEL_EXPORTER_OTLP_PROTOCOL", "OTEL_ENDPOINT", "OTEL_EXPORTER_OTLP_ENDPOINT"):
|
|
monkeypatch.delenv(var, raising=False)
|
|
monkeypatch.setenv("OTEL_EXPORTER_OTLP_TRACES_ENDPOINT", f"{base_url}/custom/traces")
|
|
|
|
cfg = OpenTelemetryV2Config.from_env()
|
|
assert cfg.exporter == "otlp_http"
|
|
_export_one_span(cfg)
|
|
assert received_paths == ["/custom/traces"]
|
|
|
|
|
|
def test_traces_endpoint_per_exporter_coexists_with_default_normalization(otlp_collector):
|
|
base_url, received_paths = otlp_collector
|
|
cfg = OpenTelemetryV2Config(
|
|
exporters=[
|
|
{"kind": "otlp_http", "endpoint": base_url},
|
|
{
|
|
"kind": "otlp_http",
|
|
"endpoint": f"{base_url}/services/collector",
|
|
"traces_endpoint": f"{base_url}/services/collector/traces",
|
|
},
|
|
]
|
|
)
|
|
_export_one_span(cfg)
|
|
assert sorted(received_paths) == ["/services/collector/traces", "/v1/traces"]
|
|
|
|
|
|
def test_http_json_exporter_honors_traces_endpoint(otlp_collector):
|
|
base_url, received_paths = otlp_collector
|
|
cfg = OpenTelemetryV2Config(
|
|
exporters=[
|
|
{
|
|
"kind": "http/json",
|
|
"endpoint": base_url,
|
|
"traces_endpoint": f"{base_url}/services/collector/traces",
|
|
}
|
|
]
|
|
)
|
|
_export_one_span(cfg)
|
|
assert received_paths == ["/services/collector/traces"]
|
|
|
|
|
|
def test_otlp_metric_exporter_uses_cumulative_histogram_temporality():
|
|
"""Histograms must export as cumulative, not delta.
|
|
|
|
Prometheus-backed OTLP receivers (Grafana Cloud / Mimir) reject delta
|
|
histograms with ``invalid temporality and type combination`` and drop the
|
|
entire metric batch, so a delta default silently loses every GenAI metric.
|
|
"""
|
|
from opentelemetry.sdk.metrics import Histogram
|
|
from opentelemetry.sdk.metrics.export import AggregationTemporality
|
|
|
|
reader = providers.build_metric_reader(OpenTelemetryV2Config(exporter="otlp_http", endpoint="http://h:4318"))
|
|
temporality = reader._exporter._preferred_temporality # noqa: SLF001 # exporter exposes no public accessor
|
|
|
|
assert temporality[Histogram] is AggregationTemporality.CUMULATIVE
|
|
|
|
|
|
def test_otlp_logs_endpoint_normalization():
|
|
norm = providers._otlp_logs_endpoint
|
|
# A base endpoint gets the signal path appended (the common OTLP env shape).
|
|
assert norm("http://collector:4318") == "http://collector:4318/v1/logs"
|
|
assert norm("http://collector:4318/") == "http://collector:4318/v1/logs"
|
|
# An already-correct path is left intact.
|
|
assert norm("http://collector:4318/v1/logs") == "http://collector:4318/v1/logs"
|
|
# A sibling signal's path is rewritten to logs, so one OTEL_ENDPOINT works
|
|
# for every signal rather than POSTing events at the traces path.
|
|
assert norm("http://collector:4318/v1/traces") == "http://collector:4318/v1/logs"
|
|
assert norm("http://collector:4318/v1/metrics") == "http://collector:4318/v1/logs"
|
|
assert norm(None) is None
|
|
|
|
|
|
def test_build_log_exporter_variants():
|
|
from opentelemetry.sdk._logs.export import ConsoleLogExporter, InMemoryLogExporter
|
|
|
|
assert isinstance(
|
|
providers.build_log_exporter(OpenTelemetryV2Config(exporter="console")),
|
|
ConsoleLogExporter,
|
|
)
|
|
assert isinstance(
|
|
providers.build_log_exporter(OpenTelemetryV2Config(exporter="in_memory")),
|
|
InMemoryLogExporter,
|
|
)
|
|
# An unrecognized kind falls back to console rather than dropping events.
|
|
assert isinstance(
|
|
providers.build_log_exporter(OpenTelemetryV2Config(exporter="unknown")),
|
|
ConsoleLogExporter,
|
|
)
|
|
http_exporter = providers.build_log_exporter(OpenTelemetryV2Config(exporter="otlp_http", endpoint="http://h:4318"))
|
|
assert "OTLPLogExporter" in type(http_exporter).__name__
|
|
|
|
|
|
def test_build_logger_provider_picks_processor_by_exporter_kind():
|
|
"""Console and in-memory exporters export synchronously (tests depend on it);
|
|
every other destination gets the batch processor."""
|
|
from opentelemetry.sdk._logs.export import (
|
|
BatchLogRecordProcessor,
|
|
ConsoleLogExporter,
|
|
InMemoryLogExporter,
|
|
SimpleLogRecordProcessor,
|
|
)
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory")
|
|
|
|
def processor_of(provider):
|
|
return provider._multi_log_record_processor._log_record_processors[0]
|
|
|
|
assert isinstance(
|
|
processor_of(providers.build_logger_provider(cfg, log_exporter=InMemoryLogExporter())),
|
|
SimpleLogRecordProcessor,
|
|
)
|
|
assert isinstance(
|
|
processor_of(providers.build_logger_provider(cfg, log_exporter=ConsoleLogExporter())),
|
|
SimpleLogRecordProcessor,
|
|
)
|
|
http_exporter = providers.build_log_exporter(OpenTelemetryV2Config(exporter="otlp_http", endpoint="http://h:4318"))
|
|
assert isinstance(
|
|
processor_of(providers.build_logger_provider(cfg, log_exporter=http_exporter)),
|
|
BatchLogRecordProcessor,
|
|
)
|
|
grpc_exporter = providers.build_span_exporter(OpenTelemetryV2Config(exporter="otlp_grpc", endpoint="http://h:4317"))
|
|
assert "OTLPSpanExporter" in type(grpc_exporter).__name__
|
|
|
|
|
|
def test_build_resource_includes_deployment_environment():
|
|
resource = providers.build_resource(OpenTelemetryV2Config(service_name="svc", deployment_environment="prod"))
|
|
assert resource.attributes["service.name"] == "svc"
|
|
assert resource.attributes["deployment.environment"] == "prod"
|
|
|
|
|
|
def test_build_tracer_provider_processor_selection():
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory")
|
|
simple = providers.build_tracer_provider(cfg, exporter=InMemorySpanExporter())
|
|
batch = providers.build_tracer_provider(cfg, exporter=ConsoleSpanExporter(), use_simple_processor=False)
|
|
# both build without error; assert the requested processor type was used
|
|
simple_procs = simple._active_span_processor._span_processors
|
|
batch_procs = batch._active_span_processor._span_processors
|
|
assert any(isinstance(p, SimpleSpanProcessor) for p in simple_procs)
|
|
assert any(isinstance(p, BatchSpanProcessor) for p in batch_procs)
|
|
|
|
|
|
def test_baggage_processor_lifecycle_noops():
|
|
proc = providers.LiteLLMBaggageSpanProcessor(allowed_keys=["litellm.team.id"])
|
|
# no-op lifecycle hooks must not raise
|
|
assert proc.on_end(None) is None # type: ignore[arg-type]
|
|
assert proc.shutdown() is None
|
|
assert proc.force_flush() is True
|
|
|
|
|
|
def test_emitter_without_call_id_is_not_deduped():
|
|
from litellm.integrations.otel.emitter import SpanEmitter
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory")
|
|
provider, exporter = providers.in_memory_provider(cfg)
|
|
engine = SpanEmitter(providers.get_tracer(provider, "t"), cfg)
|
|
data = LLMCallSpanData(
|
|
operation=GenAIOperation.CHAT,
|
|
provider="openai",
|
|
request_model="gpt-4o",
|
|
response_model=None,
|
|
response_id=None,
|
|
request_params=LLMRequestParams(),
|
|
usage=LLMUsage(),
|
|
finish_reasons=(),
|
|
error=SpanError(error_type="X", message=None),
|
|
response_cost=None,
|
|
server=None,
|
|
identity=RequestIdentity(call_id=None),
|
|
)
|
|
engine.emit(SpanRole.LLM_CALL, data)
|
|
engine.emit(SpanRole.LLM_CALL, data) # no call_id -> not deduped
|
|
assert len(exporter.get_finished_spans()) == 2
|
|
|
|
|
|
def _emit_error_span(message, error_type="litellm.APIError"):
|
|
from litellm.integrations.otel.emitter import SpanEmitter
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory")
|
|
provider, exporter = providers.in_memory_provider(cfg)
|
|
engine = SpanEmitter(providers.get_tracer(provider, "t"), cfg)
|
|
data = LLMCallSpanData(
|
|
operation=GenAIOperation.CHAT,
|
|
provider="openai",
|
|
request_model="gpt-4o",
|
|
response_model=None,
|
|
response_id=None,
|
|
request_params=LLMRequestParams(),
|
|
usage=LLMUsage(),
|
|
finish_reasons=(),
|
|
error=SpanError(error_type=error_type, message=message),
|
|
response_cost=None,
|
|
server=None,
|
|
identity=RequestIdentity(call_id=None),
|
|
)
|
|
engine.emit(SpanRole.LLM_CALL, data)
|
|
(span,) = exporter.get_finished_spans()
|
|
return span
|
|
|
|
|
|
def _exception_event(span):
|
|
from litellm.integrations.otel.model.semconv import ExceptionEvent
|
|
|
|
events = [e for e in span.events if e.name == ExceptionEvent.NAME]
|
|
assert len(events) == 1, "expected exactly one exception event"
|
|
return events[0]
|
|
|
|
|
|
def test_error_message_recorded_as_full_exception_event_untruncated():
|
|
"""The ``exception`` event carries the full untruncated message under
|
|
``exception.message`` so backends that dynamic-map unknown string span
|
|
attrs to ``keyword`` (e.g. Elasticsearch with a 1024-char ``ignore_above``)
|
|
still see it in full via the semconv-recognized event field."""
|
|
from litellm.integrations.otel.model.semconv import Error, ExceptionEvent
|
|
|
|
long_message = "boom: " + "x" * 5000
|
|
span = _emit_error_span(long_message, error_type="litellm.APIError")
|
|
|
|
event = _exception_event(span)
|
|
assert event.attributes[ExceptionEvent.MESSAGE] == long_message
|
|
assert len(event.attributes[ExceptionEvent.MESSAGE]) == len(long_message) > 1024
|
|
assert event.attributes[ExceptionEvent.TYPE] == "litellm.APIError"
|
|
|
|
# error.type stays a low-cardinality attribute; the exception EVENT field
|
|
# ``exception.message`` never becomes a bare string attribute.
|
|
assert span.attributes[Error.TYPE] == "litellm.APIError"
|
|
assert ExceptionEvent.MESSAGE not in span.attributes
|
|
assert span.status.description == long_message
|
|
|
|
|
|
def test_error_details_stamped_as_span_attributes_for_labels_ingest():
|
|
"""OTel-defined keys and litellm-specific detail keys both ride span
|
|
attributes so backends that flatten attrs into label indexes (Elastic APM
|
|
``labels.*``, Datadog span tags) render them. The exception event with the
|
|
full untruncated message stays alongside."""
|
|
from litellm.integrations.otel.model.semconv import Error, ExceptionEvent, LiteLLMError
|
|
from litellm.integrations.otel.emitter import SpanEmitter
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory")
|
|
provider, exporter = providers.in_memory_provider(cfg)
|
|
engine = SpanEmitter(providers.get_tracer(provider, "t"), cfg)
|
|
data = LLMCallSpanData(
|
|
operation=GenAIOperation.CHAT,
|
|
provider="openai",
|
|
request_model="gpt-4o",
|
|
response_model=None,
|
|
response_id=None,
|
|
request_params=LLMRequestParams(),
|
|
usage=LLMUsage(),
|
|
finish_reasons=(),
|
|
error=SpanError(
|
|
error_type="litellm.BadRequestError",
|
|
message="400: violated moderation policy",
|
|
code="400",
|
|
stack_trace="File proxy_server.py line 8570 ...",
|
|
llm_provider="openai",
|
|
),
|
|
response_cost=None,
|
|
server=None,
|
|
identity=RequestIdentity(call_id=None),
|
|
)
|
|
engine.emit(SpanRole.LLM_CALL, data)
|
|
(span,) = exporter.get_finished_spans()
|
|
|
|
# OTel-defined keys (from the ``error.*`` semconv registry).
|
|
assert span.attributes[Error.TYPE] == "litellm.BadRequestError"
|
|
assert span.attributes[Error.MESSAGE] == "400: violated moderation policy"
|
|
# LiteLLM-specific detail keys, under the ``litellm.provider.error.*``
|
|
# vendor namespace, not defined by OTel semconv.
|
|
assert span.attributes[LiteLLMError.CODE] == "400"
|
|
assert span.attributes[LiteLLMError.STACK_TRACE] == "File proxy_server.py line 8570 ..."
|
|
assert span.attributes[LiteLLMError.LLM_PROVIDER] == "openai"
|
|
|
|
# The exception event carries the same message on the span too.
|
|
event = _exception_event(span)
|
|
assert event.attributes[ExceptionEvent.MESSAGE] == "400: violated moderation policy"
|
|
|
|
|
|
def test_error_details_omitted_when_span_error_carries_only_message():
|
|
"""A guardrail-shape error (message only, no code/traceback/provider) must
|
|
not pollute the span with empty-string detail attributes. Only the keys
|
|
that carry real data land."""
|
|
from litellm.integrations.otel.model.semconv import Error, LiteLLMError
|
|
|
|
span = _emit_error_span("guardrail rejected", error_type="ContentFilter")
|
|
|
|
assert span.attributes[Error.TYPE] == "ContentFilter"
|
|
assert span.attributes[Error.MESSAGE] == "guardrail rejected"
|
|
# LiteLLM-specific detail keys aren't stamped when the SpanError doesn't
|
|
# carry them.
|
|
assert LiteLLMError.CODE not in span.attributes
|
|
assert LiteLLMError.STACK_TRACE not in span.attributes
|
|
assert LiteLLMError.LLM_PROVIDER not in span.attributes
|
|
|
|
|
|
def test_error_attribute_keys_are_pinned():
|
|
"""``error.type`` and ``error.message`` come from the semconv ``error.*``
|
|
registry; the litellm-specific detail keys are vendor keys under
|
|
``litellm.provider.error.*``. Pins the exact strings so the emitted
|
|
vocabulary can't drift silently."""
|
|
from litellm.integrations.otel.model.semconv import Error, LiteLLMError
|
|
|
|
assert Error.TYPE == "error.type"
|
|
assert Error.MESSAGE == "error.message"
|
|
assert LiteLLMError.CODE == "litellm.provider.error.code"
|
|
assert LiteLLMError.STACK_TRACE == "litellm.provider.error.stack_trace"
|
|
assert LiteLLMError.LLM_PROVIDER == "litellm.provider.error.llm_provider"
|
|
|
|
|
|
def test_error_message_falls_back_to_error_type_when_message_absent():
|
|
"""A ``SpanError(error_type=..., message=None)`` still renders on the span:
|
|
the resolved message is the error_type, and it lands on ``error.message``,
|
|
the exception event, and the span-status description in lockstep so a
|
|
single-source-of-truth view isn't inconsistent."""
|
|
from litellm.integrations.otel.model.semconv import Error, ExceptionEvent
|
|
|
|
span = _emit_error_span(message=None, error_type="RateLimitError")
|
|
|
|
assert span.attributes[Error.MESSAGE] == "RateLimitError"
|
|
assert _exception_event(span).attributes[ExceptionEvent.MESSAGE] == "RateLimitError"
|
|
assert span.status.description == "RateLimitError"
|
|
|
|
|
|
def test_success_span_records_no_exception_event():
|
|
from litellm.integrations.otel.emitter import SpanEmitter
|
|
from litellm.integrations.otel.model.semconv import ExceptionEvent
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory")
|
|
provider, exporter = providers.in_memory_provider(cfg)
|
|
engine = SpanEmitter(providers.get_tracer(provider, "t"), cfg)
|
|
data = LLMCallSpanData(
|
|
operation=GenAIOperation.CHAT,
|
|
provider="openai",
|
|
request_model="gpt-4o",
|
|
response_model="gpt-4o",
|
|
response_id="resp-1",
|
|
request_params=LLMRequestParams(),
|
|
usage=LLMUsage(),
|
|
finish_reasons=("stop",),
|
|
error=None,
|
|
response_cost=None,
|
|
server=None,
|
|
identity=RequestIdentity(call_id=None),
|
|
)
|
|
engine.emit(SpanRole.LLM_CALL, data)
|
|
(span,) = exporter.get_finished_spans()
|
|
assert all(e.name != ExceptionEvent.NAME for e in span.events)
|
|
|
|
|
|
def _engine_with_event_recorder():
|
|
from opentelemetry.sdk._logs.export import InMemoryLogExporter
|
|
|
|
from litellm.integrations.otel.emitter import SpanEmitter
|
|
from litellm.integrations.otel.plumbing.events import GenAIEventRecorder
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory", enable_events=True)
|
|
provider, span_exporter = providers.in_memory_provider(cfg)
|
|
log_exporter = InMemoryLogExporter()
|
|
logger_provider = providers.build_logger_provider(cfg, log_exporter=log_exporter)
|
|
recorder = GenAIEventRecorder(providers.get_event_logger(logger_provider))
|
|
engine = SpanEmitter(providers.get_tracer(provider, "t"), cfg, event_recorder=recorder)
|
|
return engine, span_exporter, log_exporter
|
|
|
|
|
|
def _llm_call_data(error):
|
|
return LLMCallSpanData(
|
|
operation=GenAIOperation.CHAT,
|
|
provider="openai",
|
|
request_model="gpt-4o",
|
|
response_model=None,
|
|
response_id=None,
|
|
request_params=LLMRequestParams(),
|
|
usage=LLMUsage(),
|
|
finish_reasons=(),
|
|
error=error,
|
|
response_cost=None,
|
|
server=None,
|
|
identity=RequestIdentity(call_id=None),
|
|
)
|
|
|
|
|
|
def test_operation_exception_log_event_emitted_on_failed_llm_call():
|
|
"""A failed LLM call records the GenAI semconv ``gen_ai.client.operation.exception``
|
|
event on the logs signal: severity WARN, the full ``exception.*`` trio (including
|
|
the stacktrace, which span-side only exists under a vendor key), correlated to
|
|
the failed span via trace/span ids. The span-side error surface stays intact."""
|
|
from opentelemetry._logs.severity import SeverityNumber
|
|
|
|
from litellm.integrations.otel.model.semconv import ExceptionEvent, GenAIEvent
|
|
|
|
engine, span_exporter, log_exporter = _engine_with_event_recorder()
|
|
engine.emit(
|
|
SpanRole.LLM_CALL,
|
|
_llm_call_data(
|
|
SpanError(
|
|
error_type="RateLimitError",
|
|
message="rate limited",
|
|
code="429",
|
|
stack_trace="Traceback (most recent call last) ...",
|
|
llm_provider="openai",
|
|
)
|
|
),
|
|
)
|
|
(span,) = span_exporter.get_finished_spans()
|
|
(log,) = log_exporter.get_finished_logs()
|
|
record = log.log_record
|
|
|
|
assert record.attributes["event.name"] == GenAIEvent.OPERATION_EXCEPTION
|
|
assert record.severity_number == SeverityNumber.WARN
|
|
assert record.attributes[ExceptionEvent.TYPE] == "RateLimitError"
|
|
assert record.attributes[ExceptionEvent.MESSAGE] == "rate limited"
|
|
assert record.attributes[ExceptionEvent.STACKTRACE] == "Traceback (most recent call last) ..."
|
|
assert record.trace_id == span.context.trace_id
|
|
assert record.span_id == span.context.span_id
|
|
|
|
assert [e.name for e in span.events] == [ExceptionEvent.NAME]
|
|
assert span.attributes["error.type"] == "RateLimitError"
|
|
|
|
|
|
def test_operation_exception_log_event_omits_absent_stacktrace():
|
|
from litellm.integrations.otel.model.semconv import ExceptionEvent
|
|
|
|
engine, _, log_exporter = _engine_with_event_recorder()
|
|
engine.emit(SpanRole.LLM_CALL, _llm_call_data(SpanError(error_type="APIError", message="boom")))
|
|
(log,) = log_exporter.get_finished_logs()
|
|
|
|
assert ExceptionEvent.STACKTRACE not in log.log_record.attributes
|
|
assert log.log_record.attributes[ExceptionEvent.MESSAGE] == "boom"
|
|
|
|
|
|
def test_operation_exception_log_event_always_carries_required_pair():
|
|
"""``exception.type`` and ``exception.message`` are the semconv-required pair:
|
|
they ride the event even when the recorder is handed empty strings, so an
|
|
event is never emitted with no required field. Only the stacktrace is
|
|
conditional."""
|
|
from opentelemetry.sdk._logs.export import InMemoryLogExporter
|
|
from opentelemetry.trace import INVALID_SPAN_CONTEXT
|
|
|
|
from litellm.integrations.otel.model.semconv import ExceptionEvent
|
|
from litellm.integrations.otel.plumbing.events import GenAIEventRecorder
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory", enable_events=True)
|
|
log_exporter = InMemoryLogExporter()
|
|
logger_provider = providers.build_logger_provider(cfg, log_exporter=log_exporter)
|
|
recorder = GenAIEventRecorder(providers.get_event_logger(logger_provider))
|
|
|
|
recorder.record_operation_exception(
|
|
span_context=INVALID_SPAN_CONTEXT,
|
|
error_type="",
|
|
message="",
|
|
stack_trace="",
|
|
timestamp_ns=None,
|
|
)
|
|
(log,) = log_exporter.get_finished_logs()
|
|
attributes = log.log_record.attributes
|
|
assert attributes[ExceptionEvent.TYPE] == ""
|
|
assert attributes[ExceptionEvent.MESSAGE] == ""
|
|
assert ExceptionEvent.STACKTRACE not in attributes
|
|
|
|
|
|
def test_operation_exception_log_event_records_without_the_events_api():
|
|
"""Recording must not import the Events API modules (removed upstream in 1.44.0);
|
|
the SDK record path still exports."""
|
|
import importlib
|
|
import sys
|
|
from unittest.mock import patch
|
|
|
|
from opentelemetry._logs.severity import SeverityNumber
|
|
from opentelemetry.sdk._logs.export import InMemoryLogExporter
|
|
from opentelemetry.trace import INVALID_SPAN_CONTEXT
|
|
|
|
from litellm.integrations.otel.model.semconv import ExceptionEvent, GenAIEvent
|
|
|
|
plumbing = ("litellm.integrations.otel.plumbing.events", "litellm.integrations.otel.plumbing.providers")
|
|
without_events_api = {
|
|
**{name: module for name, module in sys.modules.items() if name not in plumbing},
|
|
"opentelemetry._events": None,
|
|
"opentelemetry.sdk._events": None,
|
|
}
|
|
with patch.dict(sys.modules, without_events_api, clear=True):
|
|
events_mod = importlib.import_module(plumbing[0])
|
|
providers_mod = importlib.import_module(plumbing[1])
|
|
|
|
log_exporter = InMemoryLogExporter()
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory", enable_events=True)
|
|
logger_provider = providers_mod.build_logger_provider(cfg, log_exporter=log_exporter)
|
|
recorder = events_mod.GenAIEventRecorder(providers_mod.get_event_logger(logger_provider))
|
|
recorder.record_operation_exception(
|
|
span_context=INVALID_SPAN_CONTEXT,
|
|
error_type="RateLimitError",
|
|
message="rate limited",
|
|
stack_trace=None,
|
|
timestamp_ns=None,
|
|
)
|
|
|
|
(log,) = log_exporter.get_finished_logs()
|
|
record = log.log_record
|
|
assert record.attributes[GenAIEvent.NAME_KEY] == GenAIEvent.OPERATION_EXCEPTION
|
|
assert record.attributes[ExceptionEvent.TYPE] == "RateLimitError"
|
|
assert record.attributes[ExceptionEvent.MESSAGE] == "rate limited"
|
|
assert record.severity_number == SeverityNumber.WARN
|
|
assert record.timestamp is not None
|
|
|
|
|
|
def test_operation_exception_log_event_not_emitted_on_success():
|
|
engine, span_exporter, log_exporter = _engine_with_event_recorder()
|
|
engine.emit(SpanRole.LLM_CALL, _llm_call_data(None))
|
|
|
|
assert len(span_exporter.get_finished_spans()) == 1
|
|
assert log_exporter.get_finished_logs() == ()
|
|
|
|
|
|
def test_operation_exception_log_event_only_for_llm_call_role():
|
|
"""The event is scoped to GenAI client operations; a failed guardrail span
|
|
keeps its span-side error surface but records no GenAI exception event."""
|
|
engine, span_exporter, log_exporter = _engine_with_event_recorder()
|
|
engine.emit(
|
|
SpanRole.GUARDRAIL,
|
|
GuardrailSpanData("presidio", status="failure", error=SpanError(error_type="X", message="denied")),
|
|
)
|
|
(span,) = span_exporter.get_finished_spans()
|
|
|
|
assert span.attributes["error.type"] == "X"
|
|
assert log_exporter.get_finished_logs() == ()
|
|
|
|
|
|
def test_resolve_logger_provider_honors_explicit_noop_optout(monkeypatch):
|
|
"""A ``NoOpLoggerProvider`` global is an explicit operator opt-out from the logs
|
|
signal: resolve to ``None`` so no recorder (and so no event) is ever built,
|
|
rather than emitting into a provider that drops everything."""
|
|
from opentelemetry import _logs
|
|
from opentelemetry._logs import NoOpLoggerProvider
|
|
|
|
from litellm.integrations.otel.logger import OpenTelemetryV2
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory", enable_events=True)
|
|
tracer_provider, _ = providers.in_memory_provider(cfg)
|
|
monkeypatch.setattr(_logs, "get_logger_provider", lambda: NoOpLoggerProvider())
|
|
|
|
assert providers.resolve_logger_provider(cfg) is None
|
|
logger = OpenTelemetryV2(config=cfg, tracer_provider=tracer_provider)
|
|
assert logger._emitter._event_recorder is None
|
|
|
|
|
|
def test_resolve_logger_provider_reuses_operator_sdk_global(monkeypatch):
|
|
"""Events ride an operator-configured logs pipeline rather than a second one
|
|
built by litellm, so they land wherever the operator's other logs land."""
|
|
from opentelemetry import _logs
|
|
from opentelemetry.sdk._logs.export import InMemoryLogExporter
|
|
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory", enable_events=True)
|
|
operator_provider = providers.build_logger_provider(cfg, log_exporter=InMemoryLogExporter())
|
|
monkeypatch.setattr(_logs, "get_logger_provider", lambda: operator_provider)
|
|
|
|
assert providers.resolve_logger_provider(cfg) is operator_provider
|
|
|
|
|
|
def test_operation_exception_event_keys_are_pinned():
|
|
from litellm.integrations.otel.model.semconv import ExceptionEvent, GenAIEvent
|
|
|
|
assert GenAIEvent.OPERATION_EXCEPTION == "gen_ai.client.operation.exception"
|
|
assert ExceptionEvent.STACKTRACE == "exception.stacktrace"
|
|
|
|
|
|
# --- service taxonomy: which calls become spans, and of what kind ----------- #
|
|
|
|
|
|
def test_span_role_for_service_classifies_datastores_internal_and_metrics_only():
|
|
# Outbound datastores -> DB_CALL (CLIENT), with a db.system.
|
|
for name in (
|
|
"redis",
|
|
"postgres",
|
|
"batch_write_to_db",
|
|
"redis_daily_spend_update_queue",
|
|
):
|
|
assert span_role_for_service(name) is SpanRole.DB_CALL
|
|
assert db_system(name) is not None
|
|
# Genuine internal work worth a span -> SERVICE (INTERNAL).
|
|
assert span_role_for_service("reset_budget_job") is SpanRole.SERVICE
|
|
assert db_system("reset_budget_job") is None
|
|
# Framework instrumentation that duplicates a gen-AI span (or gets a live
|
|
# phase span) -> None: never emitted as a service span.
|
|
for name in ("self", "router", "proxy_pre_call", "auth"):
|
|
assert span_role_for_service(name) is None
|
|
|
|
|
|
# --- event_metadata sanitization -------------------------------------------- #
|
|
|
|
|
|
def test_sanitize_event_metadata_drops_objects_dumps_and_secrets():
|
|
from litellm.integrations.otel.model.payloads import sanitize_event_metadata
|
|
|
|
clean = sanitize_event_metadata(
|
|
{
|
|
"table_name": "combined_view", # safe primitive -> kept
|
|
"count": 3, # primitive -> kept (stringified)
|
|
"function_kwargs": {"prisma_client": object()}, # denylisted key
|
|
"function_args": (1, 2), # denylisted key
|
|
"user_api_key_auth": "blob", # 'auth' substring -> dropped
|
|
"api_key": "sk-secret", # 'api_key' substring -> dropped
|
|
"set-cookie": "x", # 'cookie' substring -> dropped
|
|
"hidden_params": "headers...", # denylisted substring
|
|
"obj": object(), # non-primitive value -> dropped
|
|
"nested": {"x": 1}, # non-primitive value -> dropped
|
|
}
|
|
)
|
|
assert clean == {"table_name": "combined_view", "count": "3"}
|
|
|
|
|
|
def test_sanitize_event_metadata_caps_value_length_and_handles_none():
|
|
from litellm.integrations.otel.model.payloads import sanitize_event_metadata
|
|
|
|
assert sanitize_event_metadata(None) == {}
|
|
big = sanitize_event_metadata({"k": "v" * 5000})
|
|
assert len(big["k"]) == 1024
|
|
|
|
|
|
def test_genai_mapper_guardrail_cost_in_spend_attr():
|
|
"""guardrail_cost_in_spend surfaces on the span so trace consumers can tell a
|
|
billed guardrail cost (already inside litellm.cost.total) from a report-only
|
|
one; absent means billed and the attribute stays off the span."""
|
|
from litellm.integrations.otel.model.semconv import LiteLLM
|
|
|
|
entry = {
|
|
"guardrail_name": "azure-shield",
|
|
"guardrail_status": "success",
|
|
"guardrail_usage": {"text_records": 1},
|
|
"guardrail_cost": 0.00038,
|
|
"guardrail_cost_in_spend": False,
|
|
}
|
|
attrs = GenAIMapper().map(GuardrailSpanData.from_logging_entry(entry))
|
|
assert attrs[LiteLLM.GUARDRAIL_COST_IN_SPEND] is False
|
|
assert LiteLLM.GUARDRAIL_COST_IN_SPEND == "litellm.guardrail.cost_in_spend"
|
|
|
|
billed = dict(entry)
|
|
del billed["guardrail_cost_in_spend"]
|
|
assert LiteLLM.GUARDRAIL_COST_IN_SPEND not in GenAIMapper().map(GuardrailSpanData.from_logging_entry(billed))
|
|
|
|
|
|
def _sampled_span_context():
|
|
from opentelemetry.trace import SpanContext, TraceFlags, TraceState
|
|
|
|
return SpanContext(
|
|
trace_id=0x0AF7651916CD43DD8448EB211C80319C,
|
|
span_id=0x00F067AA0BA902B7,
|
|
is_remote=False,
|
|
trace_flags=TraceFlags(TraceFlags.SAMPLED),
|
|
trace_state=TraceState(),
|
|
)
|
|
|
|
|
|
def test_operation_exception_log_event_exports_through_console_exporter():
|
|
"""The emitted record serializes through a real SDK exporter: the console
|
|
exporter only handles SDK-shaped records (``to_json`` plus a resource), so
|
|
an API-shaped record crashed the export under the repo's pinned OTel."""
|
|
import io
|
|
import json as json_mod
|
|
|
|
from opentelemetry.sdk._logs import LoggerProvider
|
|
from opentelemetry.sdk._logs.export import ConsoleLogExporter, SimpleLogRecordProcessor
|
|
from opentelemetry.sdk.resources import Resource
|
|
|
|
from litellm.integrations.otel.model.semconv import ExceptionEvent, GenAIEvent
|
|
from litellm.integrations.otel.plumbing.events import GenAIEventRecorder
|
|
|
|
out = io.StringIO()
|
|
logger_provider = LoggerProvider(resource=Resource.create({"service.name": "otel-event-test"}))
|
|
logger_provider.add_log_record_processor(SimpleLogRecordProcessor(ConsoleLogExporter(out=out)))
|
|
recorder = GenAIEventRecorder(providers.get_event_logger(logger_provider), logger_provider.resource)
|
|
recorder.record_operation_exception(
|
|
span_context=_sampled_span_context(),
|
|
error_type="RateLimitError",
|
|
message="rate limited",
|
|
stack_trace=None,
|
|
timestamp_ns=None,
|
|
)
|
|
|
|
exported = json_mod.loads(out.getvalue())
|
|
assert exported["attributes"][GenAIEvent.NAME_KEY] == GenAIEvent.OPERATION_EXCEPTION
|
|
assert exported["attributes"][ExceptionEvent.TYPE] == "RateLimitError"
|
|
assert exported["attributes"][ExceptionEvent.MESSAGE] == "rate limited"
|
|
assert exported["body"] == "rate limited"
|
|
assert exported["resource"]["attributes"]["service.name"] == "otel-event-test"
|
|
|
|
|
|
def test_operation_exception_log_event_encodes_for_otlp():
|
|
"""The OTLP log encoder reads ``log_record.resource`` and rejects a None
|
|
body on the pinned OTel line, so the event must encode into a real
|
|
ExportLogsServiceRequest, not only land in an in-memory exporter."""
|
|
from opentelemetry.exporter.otlp.proto.common._log_encoder import encode_logs
|
|
from opentelemetry.sdk._logs.export import InMemoryLogExporter
|
|
|
|
from litellm.integrations.otel.model.semconv import GenAIEvent
|
|
from litellm.integrations.otel.plumbing.events import GenAIEventRecorder
|
|
|
|
log_exporter = InMemoryLogExporter()
|
|
cfg = OpenTelemetryV2Config(exporter="in_memory", enable_events=True)
|
|
logger_provider = providers.build_logger_provider(cfg, log_exporter=log_exporter)
|
|
recorder = GenAIEventRecorder(providers.get_event_logger(logger_provider), logger_provider.resource)
|
|
recorder.record_operation_exception(
|
|
span_context=_sampled_span_context(),
|
|
error_type="RateLimitError",
|
|
message="rate limited",
|
|
stack_trace=None,
|
|
timestamp_ns=None,
|
|
)
|
|
|
|
request = encode_logs(log_exporter.get_finished_logs())
|
|
(resource_logs,) = request.resource_logs
|
|
(scope_logs,) = resource_logs.scope_logs
|
|
(encoded,) = scope_logs.log_records
|
|
encoded_attrs = {a.key: a.value.string_value for a in encoded.attributes}
|
|
assert encoded_attrs[GenAIEvent.NAME_KEY] == GenAIEvent.OPERATION_EXCEPTION
|
|
assert encoded.body.string_value == "rate limited"
|
|
resource_attrs = {a.key: a.value.string_value for a in resource_logs.resource.attributes}
|
|
assert resource_attrs["service.name"] == logger_provider.resource.attributes["service.name"]
|
|
|
|
|
|
|
|
|
|
def _isolate_v2_otlp_tls_env(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
for key in (
|
|
"SSL_VERIFY",
|
|
"SSL_CERT_FILE",
|
|
"OTEL_EXPORTER_OTLP_CERTIFICATE",
|
|
"OTEL_EXPORTER_OTLP_TRACES_CERTIFICATE",
|
|
"OTEL_EXPORTER_OTLP_METRICS_CERTIFICATE",
|
|
"OTEL_EXPORTER_OTLP_LOGS_CERTIFICATE",
|
|
):
|
|
monkeypatch.delenv(key, raising=False)
|
|
monkeypatch.setenv("OTEL_EXPORTER_OTLP_TIMEOUT", "2")
|
|
monkeypatch.setattr(litellm, "ssl_verify", True)
|
|
|
|
|
|
def test_v2_otlp_http_span_export_trusts_ssl_cert_file(monkeypatch: pytest.MonkeyPatch, tls_sink: TlsSink) -> None:
|
|
_isolate_v2_otlp_tls_env(monkeypatch)
|
|
monkeypatch.setenv("SSL_CERT_FILE", tls_sink.certificate_path)
|
|
cfg = OpenTelemetryV2Config(exporter="otlp_http", endpoint=tls_sink.url)
|
|
_export_one_span(cfg)
|
|
assert tls_sink.received.get(timeout=5) == "/v1/traces"
|
|
|
|
|
|
def test_v2_http_json_span_export_trusts_ssl_cert_file(monkeypatch: pytest.MonkeyPatch, tls_sink: TlsSink) -> None:
|
|
_isolate_v2_otlp_tls_env(monkeypatch)
|
|
monkeypatch.setenv("SSL_CERT_FILE", tls_sink.certificate_path)
|
|
cfg = OpenTelemetryV2Config(exporter="http/json", endpoint=tls_sink.url)
|
|
_export_one_span(cfg)
|
|
assert tls_sink.received.get(timeout=5) == "/v1/traces"
|
|
|
|
|
|
def test_v2_otlp_http_metric_export_trusts_ssl_cert_file(monkeypatch: pytest.MonkeyPatch, tls_sink: TlsSink) -> None:
|
|
_isolate_v2_otlp_tls_env(monkeypatch)
|
|
monkeypatch.setenv("SSL_CERT_FILE", tls_sink.certificate_path)
|
|
cfg = OpenTelemetryV2Config(exporter="otlp_http", endpoint=tls_sink.url)
|
|
reader = providers.build_metric_reader(cfg)
|
|
provider = MeterProvider(metric_readers=[reader])
|
|
try:
|
|
provider.get_meter("v2-tls-test").create_counter("tls_export_test").add(1)
|
|
assert provider.force_flush(), "metric flush failed"
|
|
assert tls_sink.received.get(timeout=5) == "/v1/metrics"
|
|
finally:
|
|
provider.shutdown()
|
|
|
|
|
|
def test_v2_otlp_http_log_export_trusts_ssl_cert_file(monkeypatch: pytest.MonkeyPatch, tls_sink: TlsSink) -> None:
|
|
_isolate_v2_otlp_tls_env(monkeypatch)
|
|
monkeypatch.setenv("SSL_CERT_FILE", tls_sink.certificate_path)
|
|
cfg = OpenTelemetryV2Config(exporter="otlp_http", endpoint=tls_sink.url)
|
|
exporter = providers.build_log_exporter(cfg)
|
|
try:
|
|
record = LogRecord(
|
|
timestamp=int(time.time() * 1e9),
|
|
observed_timestamp=int(time.time() * 1e9),
|
|
trace_id=0,
|
|
span_id=0,
|
|
trace_flags=TraceFlags(0),
|
|
severity_number=SeverityNumber.INFO,
|
|
body="v2-tls-test",
|
|
)
|
|
log_data = LogData(log_record=record, instrumentation_scope=InstrumentationScope("v2-tls-test"))
|
|
result = exporter.export([log_data])
|
|
assert result is LogExportResult.SUCCESS, f"log export failed: {result}"
|
|
assert tls_sink.received.get(timeout=5) == "/v1/logs"
|
|
finally:
|
|
exporter.shutdown()
|
|
|
|
|
|
def test_v2_otlp_http_export_skips_verification_when_ssl_verify_false(
|
|
monkeypatch: pytest.MonkeyPatch, tls_sink: TlsSink
|
|
) -> None:
|
|
_isolate_v2_otlp_tls_env(monkeypatch)
|
|
monkeypatch.setenv("SSL_VERIFY", "false")
|
|
cfg = OpenTelemetryV2Config(exporter="otlp_http", endpoint=tls_sink.url)
|
|
_export_one_span(cfg)
|
|
assert tls_sink.received.get(timeout=5) == "/v1/traces"
|
|
|
|
|
|
def test_v2_otlp_http_export_rejects_untrusted_collector_by_default(
|
|
monkeypatch: pytest.MonkeyPatch, tls_sink: TlsSink
|
|
) -> None:
|
|
|
|
|
|
_isolate_v2_otlp_tls_env(monkeypatch)
|
|
cfg = OpenTelemetryV2Config(exporter="otlp_http", endpoint=tls_sink.url)
|
|
provider = providers.build_tracer_provider(cfg)
|
|
provider.get_tracer("probe").start_span("probe").end()
|
|
try:
|
|
with contextlib.suppress(requests.exceptions.SSLError):
|
|
provider.force_flush()
|
|
assert tls_sink.received.empty(), "sink received a request it should never have trusted"
|
|
finally:
|
|
provider.shutdown()
|