diff --git a/litellm/integrations/otel/mappers/genai.py b/litellm/integrations/otel/mappers/genai.py index 6c61feced4d..244c793f44a 100644 --- a/litellm/integrations/otel/mappers/genai.py +++ b/litellm/integrations/otel/mappers/genai.py @@ -55,6 +55,12 @@ class GenAIMapper: GenAI.USAGE_INPUT_TOKENS: lambda d: d.usage.input_tokens, GenAI.USAGE_OUTPUT_TOKENS: lambda d: d.usage.output_tokens, Error.TYPE: lambda d: d.error.error_type if d.error else None, + LiteLLM.ERROR_RATE_LIMIT_CATEGORY: lambda d: ( + d.error.rate_limit_category if d.error else None + ), + LiteLLM.ERROR_RATE_LIMIT_TYPE: lambda d: ( + d.error.rate_limit_type if d.error else None + ), Server.ADDRESS: lambda d: d.server.address if d.server else None, Server.PORT: lambda d: d.server.port if d.server else None, LiteLLM.CALL_ID: lambda d: d.identity.call_id or None, diff --git a/litellm/integrations/otel/model/payloads.py b/litellm/integrations/otel/model/payloads.py index bbef40ba374..dcb01602bb0 100644 --- a/litellm/integrations/otel/model/payloads.py +++ b/litellm/integrations/otel/model/payloads.py @@ -95,6 +95,8 @@ class LLMUsage: class SpanError: error_type: str | None = None message: str | None = None + rate_limit_category: str | None = None + rate_limit_type: str | None = None @dataclass(frozen=True) @@ -498,6 +500,8 @@ def _parse_error(payload: "StandardLoggingPayload") -> SpanError | None: return SpanError( error_type=as_str(info.get("error_class")) or as_str(info.get("error_code")), message=as_str(info.get("error_message")) or as_str(payload.get("error_str")), + rate_limit_category=as_str(info.get("error_rate_limit_category")), + rate_limit_type=as_str(info.get("error_rate_limit_type")), ) diff --git a/litellm/integrations/otel/model/semconv.py b/litellm/integrations/otel/model/semconv.py index 7df07f30a01..eb76cae43a8 100644 --- a/litellm/integrations/otel/model/semconv.py +++ b/litellm/integrations/otel/model/semconv.py @@ -204,6 +204,15 @@ class LiteLLM: SERVICE_NAME: Final = "litellm.service.name" SERVICE_CALL_TYPE: Final = "litellm.service.call_type" PREPROCESSING_MS: Final = "litellm.preprocessing.duration_ms" + # Rate-limit error decomposition stamped on a failed span when the underlying + # error is a ``litellm.RateLimitError`` (vendor or proxy-internal). They sit + # in the vendor namespace because there is no GenAI semconv equivalent for + # "who rate-limited" / "which dimension was exceeded". Same source of truth + # as the StandardLoggingPayload.error_information.error_rate_limit_* + # custom-callback fields and the prometheus rate_limit_category / + # rate_limit_type labels — one decomposition, three observability surfaces. + ERROR_RATE_LIMIT_CATEGORY: Final = "litellm.error.rate_limit_category" + ERROR_RATE_LIMIT_TYPE: Final = "litellm.error.rate_limit_type" # The logical name of the MCP server a tool call was routed to. There is no # semconv key for an MCP server's *name* (the convention uses ``server.address`` # for its network location), so it lives under the vendor namespace. diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py b/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py index 2dbedda1ab6..5a47eaaba3c 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py @@ -117,6 +117,89 @@ def test_error_span_sets_status_and_error_type(): (span,) = exporter.get_finished_spans() assert span.status.status_code is StatusCode.ERROR assert span.attributes["error.type"] == "RateLimitError" + assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes + assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes + + +def test_rate_limit_error_span_carries_category_and_type(): + """A failure where ``error_information`` is populated by a litellm + ``RateLimitError`` (via ``litellm_logging.get_error_information``) carries + the unified ``error_rate_limit_category`` / ``error_rate_limit_type`` fields + onto the OTel span as ``litellm.error.rate_limit_category`` / + ``litellm.error.rate_limit_type``. Same source of truth as the + StandardLoggingPayload custom-callback channel and the prometheus + rate_limit_category / rate_limit_type labels — one decomposition stays in + sync across all three observability surfaces.""" + engine, exporter = _engine() + payload = _payload( + status="failure", + error_information={ + "error_class": "ProxyRateLimitError", + "error_message": "Rate limit exceeded for api_key: ...", + "error_rate_limit_category": "litellm_rate_limit", + "error_rate_limit_type": "requests", + "llm_provider": "openai", + }, + ) + engine.emit( + SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) + ) + (span,) = exporter.get_finished_spans() + assert span.status.status_code is StatusCode.ERROR + assert span.attributes["error.type"] == "ProxyRateLimitError" + assert ( + span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit" + ) + assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "requests" + + +def test_budget_exceeded_error_span_carries_budget_dimension(): + """``BudgetExceededError`` is the most common litellm-internal 429 case + (key/team/user/end-user budgets all hit the same code path). It exposes + ``rate_limit_category=litellm_rate_limit`` and ``rate_limit_type=budget`` + via duck-typed attribute reads in ``get_error_information``; the v2 OTel + span must surface those as well.""" + engine, exporter = _engine() + payload = _payload( + status="failure", + error_information={ + "error_class": "BudgetExceededError", + "error_message": "Budget has been exceeded! ...", + "error_rate_limit_category": "litellm_rate_limit", + "error_rate_limit_type": "budget", + "llm_provider": "openai", + }, + ) + engine.emit( + SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) + ) + (span,) = exporter.get_finished_spans() + assert span.attributes["error.type"] == "BudgetExceededError" + assert ( + span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit" + ) + assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "budget" + + +def test_non_rate_limit_error_span_omits_rate_limit_attrs(): + """Generic upstream failures (``BadRequestError``, ``APIError`` …) carry no + rate-limit decomposition; the ``litellm.error.rate_limit_*`` attributes + must stay absent rather than land as ``None`` strings.""" + engine, exporter = _engine() + payload = _payload( + status="failure", + error_information={ + "error_class": "BadRequestError", + "error_message": "invalid request", + }, + ) + engine.emit( + SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) + ) + (span,) = exporter.get_finished_spans() + assert span.attributes["error.type"] == "BadRequestError" + assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes + assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes def test_hierarchy_and_kinds_match_registry():