From 276714bdcb5b9a8e53098206ba17b508d9e0b16f Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 3 Jun 2026 06:02:33 +0000 Subject: [PATCH] feat(otel-v2): surface rate_limit_category + rate_limit_type on failed LLM-call spans MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit PR #28909 introduced the typed v2 OTel engine that builds spans from StandardLoggingPayload, with SpanError carrying error_type + message and the genai mapper stamping error.type onto every failed LLM-call span. This PR's earlier commits added error_rate_limit_category and error_rate_limit_type to the same StandardLoggingPayload.error_information the v2 engine reads — but neither field reached a span attribute, so v2 OTel traces stayed opaque about *why* a 429 fired (vendor vs litellm, RPM vs TPM vs concurrent vs budget vs max_iterations) even after the custom-callback and prometheus surfaces gained that decomposition. Three coupled changes: 1. semconv.py: add LiteLLM.ERROR_RATE_LIMIT_CATEGORY / LiteLLM.ERROR_RATE_LIMIT_TYPE under the litellm.* vendor namespace (no GenAI semconv equivalent exists for who-rate-limited / which-dimension). 2. payloads.py: extend SpanError with rate_limit_category + rate_limit_type, populated by _parse_error() from the same error_information.error_rate_limit_* fields the custom-callback channel and prometheus rate_limit_category / rate_limit_type labels read. Single source of truth across all three observability surfaces. 3. mappers/genai.py: stamp the two attributes on the LLM-call span when present. drop_none guarantees they stay absent (not 'None') for non-rate-limit failures so trace consumers can read them unconditionally. Three regression tests in test_otel_v2_emitter.py pin: a vendor / litellm-internal RateLimitError lands category=litellm_rate_limit + rate_limit_type=requests on the span; a BudgetExceededError lands rate_limit_type=budget; a non-rate-limit failure (BadRequestError) keeps the rate_limit_* attributes absent. Mutation-tested against reverting either the SpanError extension or the _parse_error read site — both new tests fail under either mutation. Co-authored-by: Mateo Wang --- litellm/integrations/otel/mappers/genai.py | 6 ++ litellm/integrations/otel/model/payloads.py | 4 + litellm/integrations/otel/model/semconv.py | 9 ++ .../integrations/otel/test_otel_v2_emitter.py | 83 +++++++++++++++++++ 4 files changed, 102 insertions(+) diff --git a/litellm/integrations/otel/mappers/genai.py b/litellm/integrations/otel/mappers/genai.py index 6c61feced4d..244c793f44a 100644 --- a/litellm/integrations/otel/mappers/genai.py +++ b/litellm/integrations/otel/mappers/genai.py @@ -55,6 +55,12 @@ class GenAIMapper: GenAI.USAGE_INPUT_TOKENS: lambda d: d.usage.input_tokens, GenAI.USAGE_OUTPUT_TOKENS: lambda d: d.usage.output_tokens, Error.TYPE: lambda d: d.error.error_type if d.error else None, + LiteLLM.ERROR_RATE_LIMIT_CATEGORY: lambda d: ( + d.error.rate_limit_category if d.error else None + ), + LiteLLM.ERROR_RATE_LIMIT_TYPE: lambda d: ( + d.error.rate_limit_type if d.error else None + ), Server.ADDRESS: lambda d: d.server.address if d.server else None, Server.PORT: lambda d: d.server.port if d.server else None, LiteLLM.CALL_ID: lambda d: d.identity.call_id or None, diff --git a/litellm/integrations/otel/model/payloads.py b/litellm/integrations/otel/model/payloads.py index bbef40ba374..dcb01602bb0 100644 --- a/litellm/integrations/otel/model/payloads.py +++ b/litellm/integrations/otel/model/payloads.py @@ -95,6 +95,8 @@ class LLMUsage: class SpanError: error_type: str | None = None message: str | None = None + rate_limit_category: str | None = None + rate_limit_type: str | None = None @dataclass(frozen=True) @@ -498,6 +500,8 @@ def _parse_error(payload: "StandardLoggingPayload") -> SpanError | None: return SpanError( error_type=as_str(info.get("error_class")) or as_str(info.get("error_code")), message=as_str(info.get("error_message")) or as_str(payload.get("error_str")), + rate_limit_category=as_str(info.get("error_rate_limit_category")), + rate_limit_type=as_str(info.get("error_rate_limit_type")), ) diff --git a/litellm/integrations/otel/model/semconv.py b/litellm/integrations/otel/model/semconv.py index 7df07f30a01..eb76cae43a8 100644 --- a/litellm/integrations/otel/model/semconv.py +++ b/litellm/integrations/otel/model/semconv.py @@ -204,6 +204,15 @@ class LiteLLM: SERVICE_NAME: Final = "litellm.service.name" SERVICE_CALL_TYPE: Final = "litellm.service.call_type" PREPROCESSING_MS: Final = "litellm.preprocessing.duration_ms" + # Rate-limit error decomposition stamped on a failed span when the underlying + # error is a ``litellm.RateLimitError`` (vendor or proxy-internal). They sit + # in the vendor namespace because there is no GenAI semconv equivalent for + # "who rate-limited" / "which dimension was exceeded". Same source of truth + # as the StandardLoggingPayload.error_information.error_rate_limit_* + # custom-callback fields and the prometheus rate_limit_category / + # rate_limit_type labels — one decomposition, three observability surfaces. + ERROR_RATE_LIMIT_CATEGORY: Final = "litellm.error.rate_limit_category" + ERROR_RATE_LIMIT_TYPE: Final = "litellm.error.rate_limit_type" # The logical name of the MCP server a tool call was routed to. There is no # semconv key for an MCP server's *name* (the convention uses ``server.address`` # for its network location), so it lives under the vendor namespace. diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py b/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py index 2dbedda1ab6..5a47eaaba3c 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py @@ -117,6 +117,89 @@ def test_error_span_sets_status_and_error_type(): (span,) = exporter.get_finished_spans() assert span.status.status_code is StatusCode.ERROR assert span.attributes["error.type"] == "RateLimitError" + assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes + assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes + + +def test_rate_limit_error_span_carries_category_and_type(): + """A failure where ``error_information`` is populated by a litellm + ``RateLimitError`` (via ``litellm_logging.get_error_information``) carries + the unified ``error_rate_limit_category`` / ``error_rate_limit_type`` fields + onto the OTel span as ``litellm.error.rate_limit_category`` / + ``litellm.error.rate_limit_type``. Same source of truth as the + StandardLoggingPayload custom-callback channel and the prometheus + rate_limit_category / rate_limit_type labels — one decomposition stays in + sync across all three observability surfaces.""" + engine, exporter = _engine() + payload = _payload( + status="failure", + error_information={ + "error_class": "ProxyRateLimitError", + "error_message": "Rate limit exceeded for api_key: ...", + "error_rate_limit_category": "litellm_rate_limit", + "error_rate_limit_type": "requests", + "llm_provider": "openai", + }, + ) + engine.emit( + SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) + ) + (span,) = exporter.get_finished_spans() + assert span.status.status_code is StatusCode.ERROR + assert span.attributes["error.type"] == "ProxyRateLimitError" + assert ( + span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit" + ) + assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "requests" + + +def test_budget_exceeded_error_span_carries_budget_dimension(): + """``BudgetExceededError`` is the most common litellm-internal 429 case + (key/team/user/end-user budgets all hit the same code path). It exposes + ``rate_limit_category=litellm_rate_limit`` and ``rate_limit_type=budget`` + via duck-typed attribute reads in ``get_error_information``; the v2 OTel + span must surface those as well.""" + engine, exporter = _engine() + payload = _payload( + status="failure", + error_information={ + "error_class": "BudgetExceededError", + "error_message": "Budget has been exceeded! ...", + "error_rate_limit_category": "litellm_rate_limit", + "error_rate_limit_type": "budget", + "llm_provider": "openai", + }, + ) + engine.emit( + SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) + ) + (span,) = exporter.get_finished_spans() + assert span.attributes["error.type"] == "BudgetExceededError" + assert ( + span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit" + ) + assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "budget" + + +def test_non_rate_limit_error_span_omits_rate_limit_attrs(): + """Generic upstream failures (``BadRequestError``, ``APIError`` …) carry no + rate-limit decomposition; the ``litellm.error.rate_limit_*`` attributes + must stay absent rather than land as ``None`` strings.""" + engine, exporter = _engine() + payload = _payload( + status="failure", + error_information={ + "error_class": "BadRequestError", + "error_message": "invalid request", + }, + ) + engine.emit( + SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) + ) + (span,) = exporter.get_finished_spans() + assert span.attributes["error.type"] == "BadRequestError" + assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes + assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes def test_hierarchy_and_kinds_match_registry():