mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
refactor(otel/v2): drop rate-limit decomposition from the LLM-call span
Proxy-side rate limits (litellm_rate_limit, budget, max_iterations) are rejected at the gate before any upstream call, so async_post_call_failure_hook tags the synthetic failure log with LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL and the v2 OTel logger never opens an LLM-call span for them; the litellm.error.rate_limit_category / litellm.error.rate_limit_type attributes were dead for exactly the cases they were meant to surface. The only failure that does open an LLM-call span carrying a RateLimitError is a vendor 429, where rate_limit_type is always None and the category just restates error.type=RateLimitError. The decomposition still reaches downstream consumers through StandardLoggingPayload.error_information.error_rate_limit_* and the prometheus rate_limit_category / rate_limit_type labels, both unchanged. Removes the SpanError fields, the _parse_error reads, the genai mapper attributes, the semconv keys, and the three span tests that asserted a scenario that never reaches the mapper in production.
This commit is contained in:
parent
ff2c03b3d3
commit
e04e4c2975
4 changed files with 0 additions and 102 deletions
|
|
@ -55,12 +55,6 @@ class GenAIMapper:
|
|||
GenAI.USAGE_INPUT_TOKENS: lambda d: d.usage.input_tokens,
|
||||
GenAI.USAGE_OUTPUT_TOKENS: lambda d: d.usage.output_tokens,
|
||||
Error.TYPE: lambda d: d.error.error_type if d.error else None,
|
||||
LiteLLM.ERROR_RATE_LIMIT_CATEGORY: lambda d: (
|
||||
d.error.rate_limit_category if d.error else None
|
||||
),
|
||||
LiteLLM.ERROR_RATE_LIMIT_TYPE: lambda d: (
|
||||
d.error.rate_limit_type if d.error else None
|
||||
),
|
||||
Server.ADDRESS: lambda d: d.server.address if d.server else None,
|
||||
Server.PORT: lambda d: d.server.port if d.server else None,
|
||||
LiteLLM.CALL_ID: lambda d: d.identity.call_id or None,
|
||||
|
|
|
|||
|
|
@ -95,8 +95,6 @@ class LLMUsage:
|
|||
class SpanError:
|
||||
error_type: str | None = None
|
||||
message: str | None = None
|
||||
rate_limit_category: str | None = None
|
||||
rate_limit_type: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
|
|
@ -500,8 +498,6 @@ def _parse_error(payload: "StandardLoggingPayload") -> SpanError | None:
|
|||
return SpanError(
|
||||
error_type=as_str(info.get("error_class")) or as_str(info.get("error_code")),
|
||||
message=as_str(info.get("error_message")) or as_str(payload.get("error_str")),
|
||||
rate_limit_category=as_str(info.get("error_rate_limit_category")),
|
||||
rate_limit_type=as_str(info.get("error_rate_limit_type")),
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -204,15 +204,6 @@ class LiteLLM:
|
|||
SERVICE_NAME: Final = "litellm.service.name"
|
||||
SERVICE_CALL_TYPE: Final = "litellm.service.call_type"
|
||||
PREPROCESSING_MS: Final = "litellm.preprocessing.duration_ms"
|
||||
# Rate-limit error decomposition stamped on a failed span when the underlying
|
||||
# error is a ``litellm.RateLimitError`` (vendor or proxy-internal). They sit
|
||||
# in the vendor namespace because there is no GenAI semconv equivalent for
|
||||
# "who rate-limited" / "which dimension was exceeded". Same source of truth
|
||||
# as the StandardLoggingPayload.error_information.error_rate_limit_*
|
||||
# custom-callback fields and the prometheus rate_limit_category /
|
||||
# rate_limit_type labels — one decomposition, three observability surfaces.
|
||||
ERROR_RATE_LIMIT_CATEGORY: Final = "litellm.error.rate_limit_category"
|
||||
ERROR_RATE_LIMIT_TYPE: Final = "litellm.error.rate_limit_type"
|
||||
# The logical name of the MCP server a tool call was routed to. There is no
|
||||
# semconv key for an MCP server's *name* (the convention uses ``server.address``
|
||||
# for its network location), so it lives under the vendor namespace.
|
||||
|
|
|
|||
|
|
@ -117,89 +117,6 @@ def test_error_span_sets_status_and_error_type():
|
|||
(span,) = exporter.get_finished_spans()
|
||||
assert span.status.status_code is StatusCode.ERROR
|
||||
assert span.attributes["error.type"] == "RateLimitError"
|
||||
assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes
|
||||
assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes
|
||||
|
||||
|
||||
def test_rate_limit_error_span_carries_category_and_type():
|
||||
"""A failure where ``error_information`` is populated by a litellm
|
||||
``RateLimitError`` (via ``litellm_logging.get_error_information``) carries
|
||||
the unified ``error_rate_limit_category`` / ``error_rate_limit_type`` fields
|
||||
onto the OTel span as ``litellm.error.rate_limit_category`` /
|
||||
``litellm.error.rate_limit_type``. Same source of truth as the
|
||||
StandardLoggingPayload custom-callback channel and the prometheus
|
||||
rate_limit_category / rate_limit_type labels — one decomposition stays in
|
||||
sync across all three observability surfaces."""
|
||||
engine, exporter = _engine()
|
||||
payload = _payload(
|
||||
status="failure",
|
||||
error_information={
|
||||
"error_class": "ProxyRateLimitError",
|
||||
"error_message": "Rate limit exceeded for api_key: ...",
|
||||
"error_rate_limit_category": "litellm_rate_limit",
|
||||
"error_rate_limit_type": "requests",
|
||||
"llm_provider": "openai",
|
||||
},
|
||||
)
|
||||
engine.emit(
|
||||
SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload)
|
||||
)
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert span.status.status_code is StatusCode.ERROR
|
||||
assert span.attributes["error.type"] == "ProxyRateLimitError"
|
||||
assert (
|
||||
span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit"
|
||||
)
|
||||
assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "requests"
|
||||
|
||||
|
||||
def test_budget_exceeded_error_span_carries_budget_dimension():
|
||||
"""``BudgetExceededError`` is the most common litellm-internal 429 case
|
||||
(key/team/user/end-user budgets all hit the same code path). It exposes
|
||||
``rate_limit_category=litellm_rate_limit`` and ``rate_limit_type=budget``
|
||||
via duck-typed attribute reads in ``get_error_information``; the v2 OTel
|
||||
span must surface those as well."""
|
||||
engine, exporter = _engine()
|
||||
payload = _payload(
|
||||
status="failure",
|
||||
error_information={
|
||||
"error_class": "BudgetExceededError",
|
||||
"error_message": "Budget has been exceeded! ...",
|
||||
"error_rate_limit_category": "litellm_rate_limit",
|
||||
"error_rate_limit_type": "budget",
|
||||
"llm_provider": "openai",
|
||||
},
|
||||
)
|
||||
engine.emit(
|
||||
SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload)
|
||||
)
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert span.attributes["error.type"] == "BudgetExceededError"
|
||||
assert (
|
||||
span.attributes[LiteLLM.ERROR_RATE_LIMIT_CATEGORY] == "litellm_rate_limit"
|
||||
)
|
||||
assert span.attributes[LiteLLM.ERROR_RATE_LIMIT_TYPE] == "budget"
|
||||
|
||||
|
||||
def test_non_rate_limit_error_span_omits_rate_limit_attrs():
|
||||
"""Generic upstream failures (``BadRequestError``, ``APIError`` …) carry no
|
||||
rate-limit decomposition; the ``litellm.error.rate_limit_*`` attributes
|
||||
must stay absent rather than land as ``None`` strings."""
|
||||
engine, exporter = _engine()
|
||||
payload = _payload(
|
||||
status="failure",
|
||||
error_information={
|
||||
"error_class": "BadRequestError",
|
||||
"error_message": "invalid request",
|
||||
},
|
||||
)
|
||||
engine.emit(
|
||||
SpanRole.LLM_CALL, LLMCallSpanData.from_standard_logging_payload(payload)
|
||||
)
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert span.attributes["error.type"] == "BadRequestError"
|
||||
assert LiteLLM.ERROR_RATE_LIMIT_CATEGORY not in span.attributes
|
||||
assert LiteLLM.ERROR_RATE_LIMIT_TYPE not in span.attributes
|
||||
|
||||
|
||||
def test_hierarchy_and_kinds_match_registry():
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue