diff --git a/litellm/integrations/otel/mappers/genai.py b/litellm/integrations/otel/mappers/genai.py index 244c793f44a..6c61feced4d 100644 --- a/litellm/integrations/otel/mappers/genai.py +++ b/litellm/integrations/otel/mappers/genai.py @@ -55,12 +55,6 @@ class GenAIMapper: GenAI.USAGE_INPUT_TOKENS: lambda d: d.usage.input_tokens, GenAI.USAGE_OUTPUT_TOKENS: lambda d: d.usage.output_tokens, Error.TYPE: lambda d: d.error.error_type if d.error else None, - LiteLLM.ERROR_RATE_LIMIT_CATEGORY: lambda d: ( - d.error.rate_limit_category if d.error else None - ), - LiteLLM.ERROR_RATE_LIMIT_TYPE: lambda d: ( - d.error.rate_limit_type if d.error else None - ), Server.ADDRESS: lambda d: d.server.address if d.server else None, Server.PORT: lambda d: d.server.port if d.server else None, LiteLLM.CALL_ID: lambda d: d.identity.call_id or None, diff --git a/litellm/integrations/otel/model/payloads.py b/litellm/integrations/otel/model/payloads.py index dcb01602bb0..bbef40ba374 100644 --- a/litellm/integrations/otel/model/payloads.py +++ b/litellm/integrations/otel/model/payloads.py @@ -95,8 +95,6 @@ class LLMUsage: class SpanError: error_type: str | None = None message: str | None = None - rate_limit_category: str | None = None - rate_limit_type: str | None = None @dataclass(frozen=True) @@ -500,8 +498,6 @@ def _parse_error(payload: "StandardLoggingPayload") -> SpanError | None: return SpanError( error_type=as_str(info.get("error_class")) or as_str(info.get("error_code")), message=as_str(info.get("error_message")) or as_str(payload.get("error_str")), - rate_limit_category=as_str(info.get("error_rate_limit_category")), - rate_limit_type=as_str(info.get("error_rate_limit_type")), ) diff --git a/litellm/integrations/otel/model/semconv.py b/litellm/integrations/otel/model/semconv.py index eb76cae43a8..7df07f30a01 100644 --- a/litellm/integrations/otel/model/semconv.py +++ b/litellm/integrations/otel/model/semconv.py @@ -204,15 +204,6 @@ class LiteLLM: SERVICE_NAME: Final = "litellm.service.name" SERVICE_CALL_TYPE: Final = "litellm.service.call_type" PREPROCESSING_MS: Final = "litellm.preprocessing.duration_ms" - # Rate-limit error decomposition stamped on a failed span when the underlying - # error is a ``litellm.RateLimitError`` (vendor or proxy-internal). They sit - # in the vendor namespace because there is no GenAI semconv equivalent for - # "who rate-limited" / "which dimension was exceeded". Same source of truth - # as the StandardLoggingPayload.error_information.error_rate_limit_* - # custom-callback fields and the prometheus rate_limit_category / - # rate_limit_type labels — one decomposition, three observability surfaces. - ERROR_RATE_LIMIT_CATEGORY: Final = "litellm.error.rate_limit_category" - ERROR_RATE_LIMIT_TYPE: Final = "litellm.error.rate_limit_type" # The logical name of the MCP server a tool call was routed to. There is no # semconv key for an MCP server's *name* (the convention uses ``server.address`` # for its network location), so it lives under the vendor namespace. diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py b/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py index 5a47eaaba3c..2dbedda1ab6 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_emitter.py @@ -117,89 +117,6 @@ def test_error_span_sets_status_and_error_type(): (span,) = exporter.get_finished_spans() assert span.status.status_code is StatusCode.ERROR assert span.attributes["error.type"] == "RateLimitError" - assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes - assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes - - -def test_rate_limit_error_span_carries_category_and_type(): - """A failure where ``error_information`` is populated by a litellm - ``RateLimitError`` (via ``litellm_logging.get_error_information``) carries - the unified ``error_rate_limit_category`` / ``error_rate_limit_type`` fields - onto the OTel span as ``litellm.error.rate_limit_category`` / - ``litellm.error.rate_limit_type``. Same source of truth as the - StandardLoggingPayload custom-callback channel and the prometheus - rate_limit_category / rate_limit_type labels — one decomposition stays in - sync across all three observability surfaces.""" - engine, exporter = _engine() - payload = _payload( - status="failure", - error_information={ - "error_class": "ProxyRateLimitError", - "error_message": "Rate limit exceeded for api_key: ...", - "error_rate_limit_category": "litellm_rate_limit", - "error_rate_limit_type": "requests", - "llm_provider": "openai", - }, - ) - engine.emit( - SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) - ) - (span,) = exporter.get_finished_spans() - assert span.status.status_code is StatusCode.ERROR - assert span.attributes["error.type"] == "ProxyRateLimitError" - assert ( - span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit" - ) - assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "requests" - - -def test_budget_exceeded_error_span_carries_budget_dimension(): - """``BudgetExceededError`` is the most common litellm-internal 429 case - (key/team/user/end-user budgets all hit the same code path). It exposes - ``rate_limit_category=litellm_rate_limit`` and ``rate_limit_type=budget`` - via duck-typed attribute reads in ``get_error_information``; the v2 OTel - span must surface those as well.""" - engine, exporter = _engine() - payload = _payload( - status="failure", - error_information={ - "error_class": "BudgetExceededError", - "error_message": "Budget has been exceeded! ...", - "error_rate_limit_category": "litellm_rate_limit", - "error_rate_limit_type": "budget", - "llm_provider": "openai", - }, - ) - engine.emit( - SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) - ) - (span,) = exporter.get_finished_spans() - assert span.attributes["error.type"] == "BudgetExceededError" - assert ( - span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit" - ) - assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "budget" - - -def test_non_rate_limit_error_span_omits_rate_limit_attrs(): - """Generic upstream failures (``BadRequestError``, ``APIError`` …) carry no - rate-limit decomposition; the ``litellm.error.rate_limit_*`` attributes - must stay absent rather than land as ``None`` strings.""" - engine, exporter = _engine() - payload = _payload( - status="failure", - error_information={ - "error_class": "BadRequestError", - "error_message": "invalid request", - }, - ) - engine.emit( - SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload) - ) - (span,) = exporter.get_finished_spans() - assert span.attributes["error.type"] == "BadRequestError" - assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes - assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes def test_hierarchy_and_kinds_match_registry():