refactor(otel/v2): drop rate-limit decomposition from the LLM-call span

Proxy-side rate limits (litellm_rate_limit, budget, max_iterations) are
rejected at the gate before any upstream call, so async_post_call_failure_hook
tags the synthetic failure log with LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL and the
v2 OTel logger never opens an LLM-call span for them; the
litellm.error.rate_limit_category / litellm.error.rate_limit_type attributes
were dead for exactly the cases they were meant to surface. The only failure
that does open an LLM-call span carrying a RateLimitError is a vendor 429, where
rate_limit_type is always None and the category just restates
error.type=RateLimitError.

The decomposition still reaches downstream consumers through
StandardLoggingPayload.error_information.error_rate_limit_* and the prometheus
rate_limit_category / rate_limit_type labels, both unchanged.

Removes the SpanError fields, the _parse_error reads, the genai mapper
attributes, the semconv keys, and the three span tests that asserted a scenario
that never reaches the mapper in production.
This commit is contained in:
mateo-berri 2026-06-03 18:50:52 +00:00
parent ff2c03b3d3
commit e04e4c2975
No known key found for this signature in database
4 changed files with 0 additions and 102 deletions

View file

@ -55,12 +55,6 @@ class GenAIMapper:
GenAI.USAGE_INPUT_TOKENS: lambda d: d.usage.input_tokens,
GenAI.USAGE_OUTPUT_TOKENS: lambda d: d.usage.output_tokens,
Error.TYPE: lambda d: d.error.error_type if d.error else None,
LiteLLM.ERROR_RATE_LIMIT_CATEGORY: lambda d: (
d.error.rate_limit_category if d.error else None
),
LiteLLM.ERROR_RATE_LIMIT_TYPE: lambda d: (
d.error.rate_limit_type if d.error else None
),
Server.ADDRESS: lambda d: d.server.address if d.server else None,
Server.PORT: lambda d: d.server.port if d.server else None,
LiteLLM.CALL_ID: lambda d: d.identity.call_id or None,

View file

@ -95,8 +95,6 @@ class LLMUsage:
class SpanError:
error_type: str | None = None
message: str | None = None
rate_limit_category: str | None = None
rate_limit_type: str | None = None
@dataclass(frozen=True)
@ -500,8 +498,6 @@ def _parse_error(payload: "StandardLoggingPayload") -> SpanError | None:
return SpanError(
error_type=as_str(info.get("error_class")) or as_str(info.get("error_code")),
message=as_str(info.get("error_message")) or as_str(payload.get("error_str")),
rate_limit_category=as_str(info.get("error_rate_limit_category")),
rate_limit_type=as_str(info.get("error_rate_limit_type")),
)

View file

@ -204,15 +204,6 @@ class LiteLLM:
SERVICE_NAME: Final = "litellm.service.name"
SERVICE_CALL_TYPE: Final = "litellm.service.call_type"
PREPROCESSING_MS: Final = "litellm.preprocessing.duration_ms"
# Rate-limit error decomposition stamped on a failed span when the underlying
# error is a ``litellm.RateLimitError`` (vendor or proxy-internal). They sit
# in the vendor namespace because there is no GenAI semconv equivalent for
# "who rate-limited" / "which dimension was exceeded". Same source of truth
# as the StandardLoggingPayload.error_information.error_rate_limit_*
# custom-callback fields and the prometheus rate_limit_category /
# rate_limit_type labels — one decomposition, three observability surfaces.
ERROR_RATE_LIMIT_CATEGORY: Final = "litellm.error.rate_limit_category"
ERROR_RATE_LIMIT_TYPE: Final = "litellm.error.rate_limit_type"
# The logical name of the MCP server a tool call was routed to. There is no
# semconv key for an MCP server's *name* (the convention uses ``server.address``
# for its network location), so it lives under the vendor namespace.

View file

@ -117,89 +117,6 @@ def test_error_span_sets_status_and_error_type():
(span,) = exporter.get_finished_spans()
assert span.status.status_code is StatusCode.ERROR
assert span.attributes["error.type"] == "RateLimitError"
assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes
assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes
def test_rate_limit_error_span_carries_category_and_type():
"""A failure where ``error_information`` is populated by a litellm
``RateLimitError`` (via ``litellm_logging.get_error_information``) carries
the unified ``error_rate_limit_category`` / ``error_rate_limit_type`` fields
onto the OTel span as ``litellm.error.rate_limit_category`` /
``litellm.error.rate_limit_type``. Same source of truth as the
StandardLoggingPayload custom-callback channel and the prometheus
rate_limit_category / rate_limit_type labels — one decomposition stays in
sync across all three observability surfaces."""
engine, exporter = _engine()
payload = _payload(
status="failure",
error_information={
"error_class": "ProxyRateLimitError",
"error_message": "Rate limit exceeded for api_key: ...",
"error_rate_limit_category": "litellm_rate_limit",
"error_rate_limit_type": "requests",
"llm_provider": "openai",
},
)
engine.emit(
SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload)
)
(span,) = exporter.get_finished_spans()
assert span.status.status_code is StatusCode.ERROR
assert span.attributes["error.type"] == "ProxyRateLimitError"
assert (
span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit"
)
assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "requests"
def test_budget_exceeded_error_span_carries_budget_dimension():
"""``BudgetExceededError`` is the most common litellm-internal 429 case
(key/team/user/end-user budgets all hit the same code path). It exposes
``rate_limit_category=litellm_rate_limit`` and ``rate_limit_type=budget``
via duck-typed attribute reads in ``get_error_information``; the v2 OTel
span must surface those as well."""
engine, exporter = _engine()
payload = _payload(
status="failure",
error_information={
"error_class": "BudgetExceededError",
"error_message": "Budget has been exceeded! ...",
"error_rate_limit_category": "litellm_rate_limit",
"error_rate_limit_type": "budget",
"llm_provider": "openai",
},
)
engine.emit(
SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload)
)
(span,) = exporter.get_finished_spans()
assert span.attributes["error.type"] == "BudgetExceededError"
assert (
span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit"
)
assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "budget"
def test_non_rate_limit_error_span_omits_rate_limit_attrs():
"""Generic upstream failures (``BadRequestError``, ``APIError`` …) carry no
rate-limit decomposition; the ``litellm.error.rate_limit_*`` attributes
must stay absent rather than land as ``None`` strings."""
engine, exporter = _engine()
payload = _payload(
status="failure",
error_information={
"error_class": "BadRequestError",
"error_message": "invalid request",
},
)
engine.emit(
SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload)
)
(span,) = exporter.get_finished_spans()
assert span.attributes["error.type"] == "BadRequestError"
assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes
assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes
def test_hierarchy_and_kinds_match_registry():