fix(otel): restore proxy-level error.* attributes on v2 failure spans (LIT-4179) (#33664)

* fix(otel): restore proxy-level error.* attributes on v2 failure spans (LIT-4179)

* refactor(otel): narrow v2 failure hook return type to drop fastapi import (LIT-4179)

---------

Co-authored-by: yucheng-berri <yucheng@berri.ai>
This commit is contained in:
devin-ai-integration[bot] 2026-07-18 10:52:27 -07:00 • committed by GitHub
parent 010b20072d
commit 4a297dd611
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
6 changed files with 328 additions and 24 deletions

View file

@ -72,6 +72,42 @@ def _stamp_litellm_error_attributes(span: Span, error: SpanError) -> None:
span.set_attribute(LiteLLMError.LLM_PROVIDER, error.llm_provider)
def stamp_error(
span: Span,
error: SpanError,
*,
record_event: bool = True,
set_status: bool = True,
) -> tuple[str, str] | None:
"""Stamp the full v2 error attribute set on ``span`` and return the resolved
``(error_type, message)`` pair, or ``None`` when the error carries neither a
type nor a message.
Shared by the LLM-call span (``finish_span``) and the proxy-level failure
spans (the FastAPI SERVER span and the ``auth`` phase span) so every v2 error
span carries identical keys. The semconv ``exception`` event rides alongside
the attributes so backends that map unknown string attrs to a truncated
``keyword`` (e.g. Elasticsearch's 1024-char ``ignore_above``) still see the
full untruncated message on the recognized event field. ``record_event`` and
``set_status`` are opt-outs for callers whose span lifecycle (``use_span``) or
owner (the FastAPI instrumentor) already records the event or the status.
"""
if not (error.error_type or error.message):
return None
error_type = error.error_type or "error"
message = error.message or error.error_type or "error"
_stamp_otel_error_attributes(span, error_type, message)
_stamp_litellm_error_attributes(span, error)
if set_status:
span.set_status(Status(StatusCode.ERROR, message))
if record_event:
span.add_event(
ExceptionEvent.NAME,
{ExceptionEvent.TYPE: error_type, ExceptionEvent.MESSAGE: message},
)
return error_type, message
class SpanEmitter:
def __init__(
self,
@ -212,21 +248,10 @@ class SpanEmitter:
)
else None
)
if error and (error.error_type or error.message):
error_type = error.error_type or "error"
message = error.message or error.error_type or "error"
_stamp_otel_error_attributes(span, error_type, message)
_stamp_litellm_error_attributes(span, error)
span.set_status(Status(StatusCode.ERROR, message))
# Also emit the semconv ``exception`` event so backends that
# dynamic-map unknown string span attrs to ``keyword`` (e.g.
# Elasticsearch with a 1024-char ``ignore_above``) still see the
# full untruncated message on the recognized event field.
span.add_event(
ExceptionEvent.NAME,
{ExceptionEvent.TYPE: error_type, ExceptionEvent.MESSAGE: message},
)
if self._event_recorder is not None and role is SpanRole.LLM_CALL:
if error:
stamped = stamp_error(span, error)
if stamped is not None and self._event_recorder is not None and role is SpanRole.LLM_CALL:
error_type, message = stamped
self._event_recorder.record_operation_exception(
span_context=span.get_span_context(),
error_type=error_type,

View file

@ -24,7 +24,7 @@ from litellm.integrations.otel.plumbing.context import (
set_request_baggage,
set_request_root_span,
)
from litellm.integrations.otel.emitter import SpanEmitter
from litellm.integrations.otel.emitter import SpanEmitter, stamp_error
from litellm.integrations.otel.mappers import resolve_mappers
from litellm.integrations.otel.model.metadata import (
LLMCallEvent,
@ -59,6 +59,7 @@ from litellm.integrations.otel.model.spans import SpanRole, span_role_for_servic
from litellm.integrations.otel.model.utils import to_ns
if TYPE_CHECKING:
from litellm.proxy._types import UserAPIKeyAuth
from litellm.types.utils import (
StandardLoggingGuardrailInformation,
StandardLoggingPayload,
@ -66,6 +67,33 @@ if TYPE_CHECKING:
LITELLM_TRACER_NAME = "litellm"
def _span_error_from_exception(
exception: "Exception | None",
*,
status_code: int | None = None,
traceback_str: str | None = None,
) -> SpanError:
"""A ``SpanError`` for a proxy-level failure that never produced a
``StandardLoggingPayload`` (auth / validation / malformed-body rejections),
mirroring ``_parse_error``'s field mapping so it stamps the same v2 keys a
failed LLM call does. ``status_code`` pins ``error.code`` to the real response
status, matching v1's SERVER-span behavior."""
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
info = StandardLoggingPayloadSetup.get_error_information(
original_exception=exception,
traceback_str=traceback_str,
)
return SpanError(
error_type=info.get("error_class") or info.get("error_code") or None,
message=info.get("error_message") or None,
code=str(status_code) if status_code is not None else (info.get("error_code") or None),
stack_trace=info.get("traceback") or None,
llm_provider=info.get("llm_provider") or None,
)
# Any callback whose class belongs to one of these modules is "the OTel
# callback" for proxy-global-registration purposes.
_OTEL_MODULES = (
@ -558,7 +586,12 @@ class OpenTelemetryV2(CustomLogger):
def start_phase_span(self, name: str) -> "Iterator[Span]":
span = self._emitter.start_span(SpanRole.SERVICE, name)
with use_span(span, end_on_exit=True):
yield span
try:
yield span
except Exception as exc:
if is_recordable_span(span):
stamp_error(span, _span_error_from_exception(exc), record_event=False, set_status=False)
raise
async def async_pre_call_hook(
self,
@ -573,6 +606,48 @@ class OpenTelemetryV2(CustomLogger):
)
return data
def record_error_attributes_on_span(
self,
span: "Span | None",
exception: "Exception | None",
status_code: int,
) -> None:
"""Stamp the v2 error.* attributes on the FastAPI-owned SERVER span for a
failure that dies before any LLM-call span exists (malformed body, auth /
validation rejection). Called from the proxy's global exception handler via
``_close_dangling_otel_server_span``. The instrumentor still owns the span's
status and lifecycle, so this only decorates it — never sets status, never
ends it — and emits no exception event, matching v1's SERVER-span behavior
and avoiding a duplicate of the event ``async_post_call_failure_hook`` or
the ``auth`` phase span already records."""
if span is None or not is_recordable_span(span):
return
stamp_error(
span,
_span_error_from_exception(exception, status_code=status_code),
record_event=False,
set_status=False,
)
async def async_post_call_failure_hook(
self,
request_data: dict,
original_exception: Exception,
user_api_key_dict: "UserAPIKeyAuth",
traceback_str: "str | None" = None,
) -> None:
"""Stamp error.* on the request's root SERVER span for a proxy-level
failure that never reached an LLM call (empty body rejected in the
endpoint, auth failure), so the failed request carries the same error keys
a failed LLM call does. v1's ``OpenTelemetry`` implemented this same hook;
v2 lost it when it stopped subclassing ``OpenTelemetry``, which is the
LIT-4179 regression for pre-call failures."""
span = request_root_span() or user_api_key_dict.parent_otel_span
if span is None or not is_recordable_span(span):
return None
stamp_error(span, _span_error_from_exception(original_exception, traceback_str=traceback_str))
return None
def emit_guardrail_span(self, entry: "StandardLoggingGuardrailInformation") -> None:
# Emitted by the guardrail-recording code the moment a guardrail finishes,
# not from a post-call hook — that hook does not fire on every path (a

View file

@ -1395,19 +1395,25 @@ def _close_dangling_otel_server_span(request: Request, status_code: int, exc: Op
if open_telemetry_logger is None:
return
# Under OTel V2 the FastAPI instrumentor owns the server span (parent_otel_span
# is that same span), and it records the error + ends it itself. Ending it here
# would end it early — losing the http.* attributes the instrumentor stamps on
# completion — and double-end it. Leave it to the instrumentor.
# is that same span) and ends it itself with the http.* attributes stamped on
# completion. The instrumentor only records an error when the exception reaches
# it uncaught, but these handlers swallow it into a JSONResponse, so it never
# does; stamp the error.* attributes here (without ending or re-statusing the
# span, which the instrumentor still owns) so pre-call failures carry the error
# like v1 did. Otherwise close and annotate the dangling span ourselves.
try:
from litellm.integrations.otel.model.config import is_otel_v2_enabled
if is_otel_v2_enabled():
return
v2_enabled = is_otel_v2_enabled()
except Exception:
pass
v2_enabled = False
try:
from opentelemetry.trace import Status, StatusCode
if v2_enabled:
if status_code >= 400:
open_telemetry_logger.record_error_attributes_on_span(parent_otel_span, exc, status_code)
return
open_telemetry_logger.set_response_status_code_attribute(parent_otel_span, status_code)
if status_code >= 400:
open_telemetry_logger.record_error_attributes_on_span(parent_otel_span, exc, status_code)
@ -1416,7 +1422,8 @@ def _close_dangling_otel_server_span(request: Request, status_code: int, exc: Op
except Exception as e:
verbose_proxy_logger.debug("Error closing dangling OTEL SERVER span: %s", str(e))
finally:
request.state.parent_otel_span = None
if not v2_enabled:
request.state.parent_otel_span = None
@app.exception_handler(RequestValidationError)

View file

@ -16,10 +16,12 @@ from litellm.integrations.otel import ( # noqa: E402
from litellm.integrations.otel.plumbing import context as ctx_mod # noqa: E402
from litellm.integrations.otel.plumbing import providers # noqa: E402
from litellm.integrations.otel.emitter import SpanEmitter # noqa: E402
from litellm.integrations.otel.emitter import stamp_error # noqa: E402
from litellm.integrations.otel.model.payloads import ( # noqa: E402
GuardrailSpanData,
LLMCallSpanData,
ServiceSpanData,
SpanError,
)
from litellm.integrations.otel.model.spans import SPAN_REGISTRY, SpanRole # noqa: E402
@ -155,6 +157,46 @@ def test_error_span_sets_status_and_error_type():
assert span.attributes["error.type"] == "RateLimitError"
def test_stamp_error_writes_full_attribute_set_and_event():
engine, exporter = _engine()
span = engine.start_span(SpanRole.PROXY_REQUEST, "POST /chat/completions")
result = stamp_error(
span, SpanError("ProxyException", "boom", code="401", stack_trace="tb", llm_provider="anthropic")
)
span.end()
(s,) = exporter.get_finished_spans()
assert result == ("ProxyException", "boom")
assert s.attributes["error.type"] == "ProxyException"
assert s.attributes["error.message"] == "boom"
assert s.attributes["litellm.provider.error.code"] == "401"
assert s.attributes["litellm.provider.error.stack_trace"] == "tb"
assert s.attributes["litellm.provider.error.llm_provider"] == "anthropic"
assert s.status.status_code is StatusCode.ERROR
assert [e.name for e in s.events] == ["exception"]
def test_stamp_error_opt_outs_skip_status_and_event():
engine, exporter = _engine()
span = engine.start_span(SpanRole.PROXY_REQUEST, "POST /chat/completions")
stamp_error(span, SpanError("ProxyException", "boom", code="401"), record_event=False, set_status=False)
span.end()
(s,) = exporter.get_finished_spans()
assert s.attributes["error.type"] == "ProxyException"
assert s.attributes["litellm.provider.error.code"] == "401"
assert s.status.status_code is StatusCode.UNSET
assert s.events == ()
def test_stamp_error_without_type_or_message_is_noop():
engine, exporter = _engine()
span = engine.start_span(SpanRole.PROXY_REQUEST, "POST /chat/completions")
assert stamp_error(span, SpanError()) is None
span.end()
(s,) = exporter.get_finished_spans()
assert "error.type" not in s.attributes
assert s.status.status_code is StatusCode.UNSET
def test_hierarchy_and_kinds_match_registry():
engine, exporter = _engine()
data = LLMCallSpanData.from_standard_logging_payload(_payload())

View file

@ -860,6 +860,120 @@ def test_guardrail_span_anchors_to_root_inside_active_phase_span():
assert guard.parent.span_id != auth_span.get_span_context().span_id
# --------------------------------------------------------------------------- #
# LIT-4179 — proxy-level failures that never reach an LLM call must still stamp
# the structured error.* attributes onto the request's spans, restoring the v1
# behavior v2 dropped when it stopped subclassing ``OpenTelemetry``.
# --------------------------------------------------------------------------- #
def _proxy_exc(message, code):
from litellm.proxy._types import ProxyException
return ProxyException(message=message, type="bad_request_error", param=None, code=code)
def test_async_post_call_failure_hook_stamps_error_on_root_span():
"""PATH B: an endpoint-level failure (empty body rejected before dispatch)
reaches ``async_post_call_failure_hook``; it must stamp error.* + an exception
event on the anchored request root span."""
from litellm.proxy._types import UserAPIKeyAuth
logger, exporter = _logger()
server = logger._emitter.start_span(SpanRole.PROXY_REQUEST, LITELLM_PROXY_REQUEST_SPAN_NAME)
set_request_root_span(server)
exc = _proxy_exc("litellm.BadRequestError: messages is required", 400)
result = asyncio.run(
logger.async_post_call_failure_hook(
request_data={}, original_exception=exc, user_api_key_dict=UserAPIKeyAuth()
)
)
server.end()
assert result is None
(span,) = exporter.get_finished_spans()
assert span.attributes["error.type"] == "ProxyException"
assert "messages is required" in span.attributes["error.message"]
assert span.attributes["litellm.provider.error.code"] == "400"
assert span.status.status_code is StatusCode.ERROR
assert any(e.name == "exception" for e in span.events)
def test_async_post_call_failure_hook_falls_back_to_user_api_key_parent_span():
"""With no anchor set (a path that never captured the root), the hook must fall
back to ``user_api_key_dict.parent_otel_span`` rather than dropping the error."""
from litellm.proxy._types import UserAPIKeyAuth
logger, exporter = _logger()
server = logger._emitter.start_span(SpanRole.PROXY_REQUEST, LITELLM_PROXY_REQUEST_SPAN_NAME)
asyncio.run(
logger.async_post_call_failure_hook(
request_data={},
original_exception=_proxy_exc("boom", 401),
user_api_key_dict=UserAPIKeyAuth(parent_otel_span=server),
)
)
server.end()
(span,) = exporter.get_finished_spans()
assert span.attributes["error.type"] == "ProxyException"
assert span.attributes["litellm.provider.error.code"] == "401"
def test_record_error_attributes_on_span_decorates_without_ending():
"""PATH A: a failure that dies before any LLM-call span (malformed body,
validation) is stamped onto the instrumentor-owned SERVER span. The method must
not end the span or emit a duplicate exception event, and must pin error.code
to the real response status (not the exception's own code)."""
logger, exporter = _logger()
server = logger._emitter.start_span(SpanRole.PROXY_REQUEST, LITELLM_PROXY_REQUEST_SPAN_NAME)
logger.record_error_attributes_on_span(server, _proxy_exc("Invalid JSON body", 400), 422)
assert server.is_recording()
server.end()
(span,) = exporter.get_finished_spans()
assert span.attributes["error.type"] == "ProxyException"
assert span.attributes["error.message"] == "Invalid JSON body"
assert span.attributes["litellm.provider.error.code"] == "422"
assert all(e.name != "exception" for e in span.events)
def test_record_error_attributes_on_span_ignores_below_400_and_missing_span():
logger, _ = _logger()
server = logger._emitter.start_span(SpanRole.PROXY_REQUEST, LITELLM_PROXY_REQUEST_SPAN_NAME)
logger.record_error_attributes_on_span(None, _proxy_exc("boom", 400), 400) # no span → no-op
logger.record_error_attributes_on_span(server, None, 400) # no exception → no-op
server.end()
assert "error.type" not in (server.attributes or {})
def test_start_phase_span_stamps_error_attributes_on_failure():
"""An ``auth`` phase span that dies (expired key) must carry the structured
error.* attributes, not only the exception event ``use_span`` records."""
logger, exporter = _logger()
server = logger._emitter.start_span(SpanRole.PROXY_REQUEST, LITELLM_PROXY_REQUEST_SPAN_NAME)
set_request_root_span(server)
exc = _proxy_exc("Authentication Error, ExpiredToken", 401)
with trace.use_span(server, end_on_exit=False):
with contextlib.suppress(Exception):
with logger.start_phase_span("auth /chat/completions"):
raise exc
server.end()
by_name = {s.name: s for s in exporter.get_finished_spans()}
auth = by_name["auth /chat/completions"]
assert auth.attributes["error.type"] == "ProxyException"
assert "ExpiredToken" in auth.attributes["error.message"]
assert auth.attributes["litellm.provider.error.code"] == "401"
assert auth.status.status_code is StatusCode.ERROR
assert any(e.name == "exception" for e in auth.events)
def test_start_phase_span_success_carries_no_error():
logger, exporter = _logger()
with logger.start_phase_span("auth /chat/completions"):
pass
(span,) = exporter.get_finished_spans()
assert "error.type" not in span.attributes
assert span.status.status_code is not StatusCode.ERROR
def test_real_logging_pre_call_opens_span_end_to_end():
"""Regression guard: a real ``LiteLLMLoggingObj.pre_call`` must fire
``log_pre_api_call`` on the V2 logger (via ``litellm.input_callback``), so the

View file

@ -124,6 +124,47 @@ def test_close_dangling_otel_server_span_records_status_and_ends(monkeypatch):
}
def test_close_dangling_otel_server_span_v2_stamps_error_without_ending(monkeypatch):
"""LIT-4179: under OTel v2 the FastAPI instrumentor owns the SERVER span, so
the handler must only stamp error.* on it (via record_error_attributes_on_span)
and must NOT set status, end the span, or clear request state — otherwise the
instrumentor's http.* attributes and span close are lost."""
import litellm.integrations.otel.model.config as otel_config
import litellm.proxy.proxy_server as ps
span = MagicMock()
fake_logger = MagicMock()
monkeypatch.setattr(ps, "open_telemetry_logger", fake_logger, raising=False)
monkeypatch.setattr(otel_config, "is_otel_v2_enabled", lambda: True)
request = _make_request(parent_otel_span=span)
exc = ProxyException(message="bad", type="bad_request_error", param=None, code=400)
_close_dangling_otel_server_span(request=request, status_code=422, exc=exc)
fake_logger.record_error_attributes_on_span.assert_called_once_with(span, exc, 422)
assert not span.end.called
assert not span.set_status.called
assert not fake_logger.set_response_status_code_attribute.called
assert request.state.parent_otel_span is span
def test_close_dangling_otel_server_span_v2_success_does_not_stamp(monkeypatch):
"""Under v2 a sub-400 status must not stamp an error onto the SERVER span."""
import litellm.integrations.otel.model.config as otel_config
import litellm.proxy.proxy_server as ps
span = MagicMock()
fake_logger = MagicMock()
monkeypatch.setattr(ps, "open_telemetry_logger", fake_logger, raising=False)
monkeypatch.setattr(otel_config, "is_otel_v2_enabled", lambda: True)
request = _make_request(parent_otel_span=span)
_close_dangling_otel_server_span(request=request, status_code=200)
assert not fake_logger.record_error_attributes_on_span.called
assert not span.end.called
def test_close_dangling_otel_server_span_missing_span_is_noop_error():
"""When parent_otel_span is missing the call short-circuits — no error."""
request = _make_request(parent_otel_span=None)