diff --git a/litellm/integrations/otel/emitter.py b/litellm/integrations/otel/emitter.py index ccbeea7cbdd..88cbf4f2488 100644 --- a/litellm/integrations/otel/emitter.py +++ b/litellm/integrations/otel/emitter.py @@ -222,6 +222,7 @@ class SpanEmitter: start_time_ns: int | None = None, end_time_ns: int | None = None, tracers: Sequence[Tracer], + links: Sequence[Link] | None = None, ) -> Span | None: """Emit one logical span once per tracer, deduping the call ONCE. @@ -244,6 +245,7 @@ class SpanEmitter: parent_context=parent_context, start_time_ns=start_time_ns, tracer=tracer, + links=links, ) self.finish_span(role, span, data, end_time_ns=end_time_ns) if first is None: diff --git a/litellm/integrations/otel/logger.py b/litellm/integrations/otel/logger.py index 5f96fe6938a..9fc268d90fe 100644 --- a/litellm/integrations/otel/logger.py +++ b/litellm/integrations/otel/logger.py @@ -242,6 +242,11 @@ class OpenTelemetryV2(CustomLogger): # call id; keep the first span so its start time is the true one. if call_id in self._open_llm_calls: return + # A retry that reuses a call id whose span already closed must not open new spans: + # the close callback short-circuits on ``_closed_call_ids``, so those spans would + # never be finished or exported (a leak). The completed attempt was already traced. + if call_id in self._closed_call_ids: + return start_time_ns = to_ns(datetime.now()) spans: tuple[Span, ...] = () # Parent to the anchored root span (ambient on the SDK path); open live only when that @@ -354,12 +359,20 @@ class OpenTelemetryV2(CustomLogger): self._open_llm_calls.pop(data.identity.call_id, None) parent_context, links = resolve_mcp_span_context() parent_context = self._seed_identity_baggage(data.identity, None, parent_context) - self._emitter.emit( + # The tool-call span carries ``gen_ai.operation.name`` (execute_tool), so the + # fan-out processor treats it as a gen-AI span and skips it; route it to the + # request's admin destinations the same way the LLM-call span is, or it would + # reach the global exporter only and be missing from every tenant destination. + call = LLMCallEvent.from_dict(kwargs) + self._emitter.emit_fanout( SpanRole.MCP_TOOL_CALL, data, parent_context=parent_context, start_time_ns=to_ns(start_time), end_time_ns=to_ns(end_time), + tracers=self._tenant_tracers.genai_tracers_for( + self.tracer, self._destinations_for_backend(call), call.dynamic_params + ), links=links, ) return True diff --git a/litellm/integrations/otel/model/config.py b/litellm/integrations/otel/model/config.py index 9ff6db2566d..dddb8d619bf 100644 --- a/litellm/integrations/otel/model/config.py +++ b/litellm/integrations/otel/model/config.py @@ -244,14 +244,16 @@ class OpenTelemetryV2Config(BaseSettings): if self.endpoint and self.exporter == "console": self.exporter = "otlp_http" # When no explicit destinations are given, fold the single-destination - # shorthand into one spec so the provider has a destination. A bare config - # whose only shorthand is the default console kind with no endpoint is the - # "nothing configured" degrade case (a preset returned no credentials, or v2 - # is enabled with no exporter set); leave it exporter-less so the provider - # exports nothing, rather than folding it into a console exporter that prints - # every span -- including prompt and completion content -- to stdout - # synchronously on the request path. - if not self.exporters and (self.endpoint or self.exporter != "console"): + # shorthand into one spec so the provider has a destination. The exception is + # the "nothing configured" degrade case -- a preset returning a bare config + # because it found no credentials, where ``exporter`` is left at its default + # ``console`` and no endpoint is set: leave it exporter-less so the provider + # exports nothing, rather than degrading to a console exporter that prints every + # span (including prompt and completion content) to stdout synchronously on the + # request path. An explicitly chosen exporter (even ``console``), a non-console + # kind, or an endpoint all still fold. + console_by_default = self.exporter == "console" and "exporter" not in self.model_fields_set + if not self.exporters and (self.endpoint or not console_by_default): self.exporters = [ ExporterSpec( kind=self.exporter, diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index d880b9b9d8e..2ca1f10687f 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -672,7 +672,12 @@ async def _resolve_logging_exporters( backend = (credential.credential_info or {}).get("description") if not backend: return None - values = {str(key): str(value) for key, value in (credential.credential_values or {}).items()} + # Drop unset (``None``) values rather than stringifying them: ``str(None)`` is the + # literal ``"None"``, which would land in the exporter endpoint/headers and break + # the export (e.g. an empty ``otel_endpoint`` becoming the URL ``"None"``). + values = { + str(key): str(value) for key, value in (credential.credential_values or {}).items() if value is not None + } destination = build_destination(backend, values) return None if destination is None else (backend, destination) diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_logger.py b/tests/test_litellm/integrations/otel/test_otel_v2_logger.py index 9693950e834..2400f3a741e 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_logger.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_logger.py @@ -449,6 +449,48 @@ def test_mcp_tool_call_is_not_logged_as_llm_call(): assert "gen_ai.request.model" not in span.attributes +def test_mcp_tool_call_routes_to_admin_destination(): + """The MCP tool-call span carries ``gen_ai.operation.name`` (execute_tool), so the + fan-out processor skips it; it must still be routed to the request's admin + destinations like the LLM-call span, not reach the global exporter only.""" + from opentelemetry.sdk.trace.export.in_memory_span_exporter import ( + InMemorySpanExporter, + ) + + from litellm.integrations.otel.model.config import ExporterSpec, OpenTelemetryV2Config + from litellm.integrations.otel.model.destination import OtelDestination + from litellm.integrations.otel.plumbing.context import _request_destinations + from litellm.integrations.otel.plumbing.providers import build_tracer_provider + + cfg = OpenTelemetryV2Config( + service_name="litellm-proxy", + exporters=[ExporterSpec(kind="in_memory")], + ) + logger = OpenTelemetryV2(config=cfg, callback_name="generic", tracer_provider=build_tracer_provider(cfg)) + dest = OtelDestination(callback_name="generic", endpoint="http://collector/v1/traces", headers={"x": "1"}) + + token = _request_destinations.set((dest,)) + try: + kwargs = {"standard_logging_object": _mcp_payload()} + asyncio.run(logger.async_log_success_event(kwargs, None, None, None)) + finally: + _request_destinations.reset(token) + + # The destination's clone provider (base in_memory exporter + the destination) must + # have exported the tool-call span. + assert logger._tenant_tracers._providers, "no destination provider was built for the tool-call" + captured = [] + for provider in logger._tenant_tracers._providers.values(): + provider.force_flush() + for proc in provider._active_span_processor._span_processors: + exporter = getattr(proc, "span_exporter", None) + if isinstance(exporter, InMemorySpanExporter): + captured += exporter.get_finished_spans() + assert any(s.attributes.get("mcp.method.name") == "tools/call" for s in captured), ( + "tool-call span did not reach the admin destination's provider" + ) + + def test_mcp_tool_call_captures_io_when_enabled(): logger, exporter = _logger_capturing() kwargs = {"standard_logging_object": _mcp_payload()} @@ -805,6 +847,28 @@ def test_pre_call_idempotent_keeps_first_span(): assert first is second # not overwritten +def test_pre_call_after_close_does_not_reopen_leaked_span(): + """A retry that reuses a call id whose span already closed must NOT open a new + span: the close callback short-circuits on ``_closed_call_ids``, so a reopened + span would never be finished or exported (a leak).""" + logger, _ = _logger() + kwargs = _kwargs() + server = logger._emitter.start_span(SpanRole.PROXY_REQUEST, LITELLM_PROXY_REQUEST_SPAN_NAME) + with trace.use_span(server, end_on_exit=False): + logger.log_pre_api_call(model="gpt-4o", messages=[], kwargs=kwargs) + assert "call_1" in logger._open_llm_calls + + asyncio.run(logger.async_log_success_event(kwargs, None, None, None)) + assert "call_1" not in logger._open_llm_calls + assert "call_1" in logger._closed_call_ids + + # A retry reuses the same call id; pre_call must not reopen a carrier/span for it. + with trace.use_span(server, end_on_exit=False): + logger.log_pre_api_call(model="gpt-4o", messages=[], kwargs=kwargs) + server.end() + assert "call_1" not in logger._open_llm_calls + + # --------------------------------------------------------------------------- # # Parent resolution — ambient context at the boundary (no metadata threading) # --------------------------------------------------------------------------- # diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_presets.py b/tests/test_litellm/integrations/otel/test_otel_v2_presets.py index e88580aaa6e..4ebd4f2bcba 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_presets.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_presets.py @@ -292,6 +292,10 @@ def test_bare_and_degraded_configs_do_not_console_flood(monkeypatch): assert OpenTelemetryV2Config().exporters == [] # An explicit endpoint (a real destination) still folds into one OTLP exporter. assert len(OpenTelemetryV2Config(endpoint="http://collector/v1/traces").exporters) == 1 + # A deployment that *explicitly* selects console output still gets it -- only the + # default/degrade console (exporter left unset) is suppressed. + explicit_console = OpenTelemetryV2Config(exporter="console") + assert len(explicit_console.exporters) == 1 and explicit_console.exporters[0].kind == "console" for var in ("LANGFUSE_PUBLIC_KEY", "LANGFUSE_SECRET_KEY", "WANDB_API_KEY", "WANDB_PROJECT_ID"): monkeypatch.delenv(var, raising=False)