From e04e4c2975f8e1076fb8833895b7b6213edee7df Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 3 Jun 2026 18:50:52 +0000 Subject: [PATCH] refactor(otel/v2): drop rate-limit decomposition from the LLM-call span Proxy-side rate limits (litellm_rate_limit, budget, max_iterations) are rejected at the gate before any upstream call, so async_post_call_failure_hook tags the synthetic failure log with LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL and the v2 OTel logger never opens an LLM-call span for them; the litellm.error.rate_limit_category / litellm.error.rate_limit_type attributes were dead for exactly the cases they were meant to surface. The only failure that does open an LLM-call span carrying a RateLimitError is a vendor 429, where rate_limit_type is always None and the category just restates error.type=RateLimitError. The decomposition still reaches downstream consumers through StandardLoggingPayload.error_information.error_rate_limit_* and the prometheus rate_limit_category / rate_limit_type labels, both unchanged. Removes the SpanError fields, the _parse_error reads, the genai mapper attributes, the semconv keys, and the three span tests that asserted a scenario that never reaches the mapper in production. --- litellm/integrations/otel/mappers/genai.py | 6 -- litellm/integrations/otel/model/payloads.py | 4 - litellm/integrations/otel/model/semconv.py | 9 -- .../integrations/otel/test_otel_v2_emitter.py | 83 ------------------- 4 files changed, 102 deletions(-) diff --git a/litellm/integrations/otel/mappers/genai.py b/litellm/integrations/otel/mappers/genai.py index 244c793f44a..6c61feced4d 100644 --- a/litellm/integrations/otel/mappers/genai.py +++ b/litellm/integrations/otel/mappers/genai.py @@ -55,12 +55,6 @@ class GenAIMapper: GenAI.USAGE_INPUT_TOKENS: lambda d: d.usage.input_tokens, GenAI.USAGE_OUTPUT_TOKENS: lambda d: d.usage.output_tokens, Error.TYPE: lambda d: d.error.error_type if d.error else None, - LiteLLM.ERROR_RATE_LIMIT_CATEGORY: lambda d: ( - d.error.rate_limit_category if d.error else None - ), - LiteLLM.ERROR_RATE_LIMIT_TYPE: lambda d: ( - d.error.rate_limit_type if d.error else None - ), Server.ADDRESS: lambda d: d.server.address if d.server else None, Server.PORT: lambda d: d.server.port if d.server else None, LiteLLM.CALL_ID: lambda d: d.identity.call_id or None, diff --git a/litellm/integrations/otel/model/payloads.py b/litellm/integrations/otel/model/payloads.py index dcb01602bb0..bbef40ba374 100644 --- a/litellm/integrations/otel/model/payloads.py +++ b/litellm/integrations/otel/model/payloads.py @@ -95,8 +95,6 @@ class LLMUsage: class SpanError: error_type: str | None = None message: str | None = None - rate_limit_category: str | None = None - rate_limit_type: str | None = None @dataclass(frozen=True) @@ -500,8 +498,6 @@ def _parse_error(payload: "StandardLoggingPayload") -> SpanError | None: return SpanError( error_type=as_str(info.get("error_class")) or as_str(info.get("error_code")), message=as_str(info.get("error_message")) or as_str(payload.get("error_str")), - rate_limit_category=as_str(info.get("error_rate_limit_category")), - rate_limit_type=as_str(info.get("error_rate_limit_type")), ) diff --git a/litellm/integrations/otel/model/semconv.py b/litellm/integrations/otel/model/semconv.py index eb76cae43a8..7df07f30a01 100644 --- a/litellm/integrations/otel/model/semconv.py +++ b/litellm/integrations/otel/model/semconv.py @@ -204,15 +204,6 @@ class LiteLLM: SERVICE_NAME: Final = "litellm.service.name" SERVICE_CALL_TYPE: Final = "litellm.service.call_type" PREPROCESSING_MS: Final = "litellm.preprocessing.duration_ms" - # Rate-limit error decomposition stamped on a failed span when the underlying - # error is a ``litellm.RateLimitError`` (vendor or proxy-internal). They sit - # in the vendor namespace because there is no GenAI semconv equivalent for - # "who rate-limited" / "which dimension was exceeded". Same source of truth - # as the StandardLoggingPayload.error_information.error_rate_limit_* - # custom-callback fields and the prometheus rate_limit_category / - # rate_limit_type labels — one decomposition, three observability surfaces. - ERROR_RATE_LIMIT_CATEGORY: Final = "litellm.error.rate_limit_category" - ERROR_RATE_LIMIT_TYPE: Final = "litellm.error.rate_limit_type" # The logical name of the MCP server a tool call was routed to. There is no # semconv key for an MCP server's *name* (the convention uses ``server.address`` # for its network location), so it lives under the vendor namespace. diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py b/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py index 5a47eaaba3c..2dbedda1ab6 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py @@ -117,89 +117,6 @@ def test_error_span_sets_status_and_error_type(): (span,) = exporter.get_finished_spans() assert span.status.status_code is StatusCode.ERROR assert span.attributes["error.type"] == "RateLimitError" - assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes - assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes - - -def test_rate_limit_error_span_carries_category_and_type(): - """A failure where ``error_information`` is populated by a litellm - ``RateLimitError`` (via ``litellm_logging.get_error_information``) carries - the unified ``error_rate_limit_category`` / ``error_rate_limit_type`` fields - onto the OTel span as ``litellm.error.rate_limit_category`` / - ``litellm.error.rate_limit_type``. Same source of truth as the - StandardLoggingPayload custom-callback channel and the prometheus - rate_limit_category / rate_limit_type labels — one decomposition stays in - sync across all three observability surfaces.""" - engine, exporter = _engine() - payload = _payload( - status="failure", - error_information={ - "error_class": "ProxyRateLimitError", - "error_message": "Rate limit exceeded for api_key: ...", - "error_rate_limit_category": "litellm_rate_limit", - "error_rate_limit_type": "requests", - "llm_provider": "openai", - }, - ) - engine.emit( - SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) - ) - (span,) = exporter.get_finished_spans() - assert span.status.status_code is StatusCode.ERROR - assert span.attributes["error.type"] == "ProxyRateLimitError" - assert ( - span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit" - ) - assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "requests" - - -def test_budget_exceeded_error_span_carries_budget_dimension(): - """``BudgetExceededError`` is the most common litellm-internal 429 case - (key/team/user/end-user budgets all hit the same code path). It exposes - ``rate_limit_category=litellm_rate_limit`` and ``rate_limit_type=budget`` - via duck-typed attribute reads in ``get_error_information``; the v2 OTel - span must surface those as well.""" - engine, exporter = _engine() - payload = _payload( - status="failure", - error_information={ - "error_class": "BudgetExceededError", - "error_message": "Budget has been exceeded! ...", - "error_rate_limit_category": "litellm_rate_limit", - "error_rate_limit_type": "budget", - "llm_provider": "openai", - }, - ) - engine.emit( - SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) - ) - (span,) = exporter.get_finished_spans() - assert span.attributes["error.type"] == "BudgetExceededError" - assert ( - span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit" - ) - assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "budget" - - -def test_non_rate_limit_error_span_omits_rate_limit_attrs(): - """Generic upstream failures (``BadRequestError``, ``APIError`` …) carry no - rate-limit decomposition; the ``litellm.error.rate_limit_*`` attributes - must stay absent rather than land as ``None`` strings.""" - engine, exporter = _engine() - payload = _payload( - status="failure", - error_information={ - "error_class": "BadRequestError", - "error_message": "invalid request", - }, - ) - engine.emit( - SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) - ) - (span,) = exporter.get_finished_spans() - assert span.attributes["error.type"] == "BadRequestError" - assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes - assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes def test_hierarchy_and_kinds_match_registry():