From 0a8ae7e624cd5b3ecc10ae871fcb244cf2901a72 Mon Sep 17 00:00:00 2001 From: jesus Date: Mon, 28 Sep 2026 22:34:52 +0000 Subject: [PATCH 1/3] fix(otel): emit V2 LLM-call span for cache hits with explicit cache attributes Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/integrations/otel/logger.py | 8 +++-- litellm/integrations/otel/mappers/genai.py | 2 ++ litellm/integrations/otel/model/metadata.py | 5 +++ litellm/integrations/otel/model/payloads.py | 4 +++ litellm/integrations/otel/model/semconv.py | 4 +++ .../otel/test_otel_v2_components.py | 13 ++++++++ .../integrations/otel/test_otel_v2_logger.py | 32 +++++++++++++++++++ 7 files changed, 66 insertions(+), 2 deletions(-) diff --git a/litellm/integrations/otel/logger.py b/litellm/integrations/otel/logger.py index e21711c2708..46d35a788a8 100644 --- a/litellm/integrations/otel/logger.py +++ b/litellm/integrations/otel/logger.py @@ -508,8 +508,12 @@ class OpenTelemetryV2(CustomLogger): # logger is a success/failure callback only, so ``pre_call`` never reaches it # and no carrier exists. The payload plus the request-level provider-handoff # stamp (``upstream_started``) is the affirmative signal of a real call; a - # gate rejection carries ``is_no_upstream_call`` and gets no span. - if carrier is None and (call.is_no_upstream_call or not call.upstream_started or call.payload is None): + # gate rejection carries ``is_no_upstream_call`` and gets no span. A cache + # hit never hands off to a provider, so ``cache_hit`` substitutes for the + # upstream stamp — the payload proves the call completed. + if carrier is None and ( + call.is_no_upstream_call or call.payload is None or (not call.upstream_started and not call.cache_hit) + ): return None try: return self._finish_carrier(carrier, call, start_time, end_time) diff --git a/litellm/integrations/otel/mappers/genai.py b/litellm/integrations/otel/mappers/genai.py index e37da8908e4..8342bd8c0f8 100644 --- a/litellm/integrations/otel/mappers/genai.py +++ b/litellm/integrations/otel/mappers/genai.py @@ -89,6 +89,8 @@ class GenAIMapper: f"{LiteLLM.COST_PREFIX}margin_fixed_amount": lambda d: d.cost.margin_fixed_amount, f"{LiteLLM.COST_PREFIX}margin_percent": lambda d: d.cost.margin_percent, f"{LiteLLM.COST_PREFIX}margin_total_amount": lambda d: d.cost.margin_total_amount, + LiteLLM.CACHE_HIT: lambda d: d.cache_hit, + LiteLLM.SAVED_CACHE_COST: lambda d: d.saved_cache_cost, LiteLLM.REQUEST_STREAMING: lambda d: d.is_streaming, LiteLLM.REQUEST_ROUTE: lambda d: d.request_route, } diff --git a/litellm/integrations/otel/model/metadata.py b/litellm/integrations/otel/model/metadata.py index ede8ac99467..6f2699335a0 100644 --- a/litellm/integrations/otel/model/metadata.py +++ b/litellm/integrations/otel/model/metadata.py @@ -220,6 +220,10 @@ class LLMCallEvent: # actually attempted — router pre-call rejections, SDK failures before the # provider handoff, and standalone guardrail runs all lack it. upstream_started: bool + # True for a response served from the litellm response cache. A cache hit + # never hands off to a provider, so ``upstream_started`` stays False even + # though the call completed and is worth a span. + cache_hit: bool # A best-effort ``"{operation} {model}"`` name known at ``pre_call`` time. The # span is renamed from the typed payload at close (``finish_span``); this only # needs to be reasonable for a span that never gets closed (a leak). @@ -242,6 +246,7 @@ class LLMCallEvent: auth_metadata=auth_metadata(payload, kwargs), is_no_upstream_call=bool(kwargs.get(LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL)), upstream_started=kwargs.get("api_call_start_time") is not None, + cache_hit=bool(cast("Mapping[str, object]", payload).get("cache_hit")) if payload else False, provisional_span_name=f"{operation.value} {model}".strip(), time_to_first_chunk_seconds=time_to_first_chunk_seconds(kwargs), trace=trace, diff --git a/litellm/integrations/otel/model/payloads.py b/litellm/integrations/otel/model/payloads.py index 7e47abfb20d..a41b0eb2562 100644 --- a/litellm/integrations/otel/model/payloads.py +++ b/litellm/integrations/otel/model/payloads.py @@ -411,6 +411,8 @@ class LLMCallSpanData: trace: TraceControls = field(default_factory=TraceControls) session_id: str | None = None embedding_output: EmbeddingOutput | None = None + cache_hit: bool | None = None + saved_cache_cost: float | None = None @classmethod def from_standard_logging_payload( @@ -469,6 +471,8 @@ class LLMCallSpanData: trace=trace or TraceControls(), session_id=session_id or None, embedding_output=embedding_output if capture_content else None, + cache_hit=as_bool(cast("Mapping[str, object]", payload).get("cache_hit")), + saved_cache_cost=as_float(cast("Mapping[str, object]", payload).get("saved_cache_cost")), ) diff --git a/litellm/integrations/otel/model/semconv.py b/litellm/integrations/otel/model/semconv.py index 19b319009e8..858c87baf07 100644 --- a/litellm/integrations/otel/model/semconv.py +++ b/litellm/integrations/otel/model/semconv.py @@ -332,6 +332,10 @@ class LiteLLM: # semconv key for an MCP server's *name* (the convention uses ``server.address`` # for its network location), so it lives under the vendor namespace. MCP_SERVER_NAME: Final = "litellm.mcp.server.name" + # Whether the response was served from the litellm response cache instead of + # an upstream provider call, and the USD cost the hit avoided. + CACHE_HIT: Final = "litellm.cache_hit" + SAVED_CACHE_COST: Final = "litellm.saved_cache_cost" class Metric: diff --git a/tests/unit/integrations/otel/test_otel_v2_components.py b/tests/unit/integrations/otel/test_otel_v2_components.py index fb7be0dda14..a6c00e1e7d9 100644 --- a/tests/unit/integrations/otel/test_otel_v2_components.py +++ b/tests/unit/integrations/otel/test_otel_v2_components.py @@ -355,6 +355,19 @@ def test_genai_mapper_cost_breakdown_absent(): assert not any(k.startswith(LiteLLM.COST_PREFIX) and k != f"{LiteLLM.COST_PREFIX}total" for k in attrs) +def test_genai_mapper_cache_hit_attrs(): + from litellm.integrations.otel.model.semconv import LiteLLM + + attrs = GenAIMapper().map(replace(_full_llm_call(), cache_hit=True, saved_cache_cost=0.002)) + assert attrs[LiteLLM.CACHE_HIT] is True + assert attrs[LiteLLM.SAVED_CACHE_COST] == 0.002 + + # The default (None) keeps the span sparse: neither key present. + uncached = GenAIMapper().map(_full_llm_call()) + assert LiteLLM.CACHE_HIT not in uncached + assert LiteLLM.SAVED_CACHE_COST not in uncached + + def test_llm_cost_from_breakdown_maps_costbreakdown_keys(): cost = LLMCost.from_breakdown( { diff --git a/tests/unit/integrations/otel/test_otel_v2_logger.py b/tests/unit/integrations/otel/test_otel_v2_logger.py index 62bf75bd083..49e1b48bc08 100644 --- a/tests/unit/integrations/otel/test_otel_v2_logger.py +++ b/tests/unit/integrations/otel/test_otel_v2_logger.py @@ -432,6 +432,38 @@ def test_no_span_when_pre_call_never_ran(): assert exporter.get_finished_spans() == () # no phantom LLM span +def test_cache_hit_without_pre_call_emits_zero_cost_span(): + """A response served from the litellm cache never hands off to a provider, + so ``pre_call`` never runs and ``api_call_start_time`` is never stamped — + but the completed call's payload still warrants a span, carrying the cache + attributes rather than being dropped like a gate rejection.""" + logger, exporter = _logger() + payload = _payload(cache_hit=True, response_cost=0.0, saved_cache_cost=0.002, cost_breakdown=None) + asyncio.run( + logger.async_log_success_event(_kwargs(payload=payload), None, None, None) + ) + (span,) = exporter.get_finished_spans() + assert span.attributes[LiteLLM.CACHE_HIT] is True + assert span.attributes[LiteLLM.SAVED_CACHE_COST] == 0.002 + assert span.attributes[f"{LiteLLM.COST_PREFIX}total"] == 0.0 + assert not [ + k + for k in span.attributes + if k.startswith(LiteLLM.COST_PREFIX) and k != f"{LiteLLM.COST_PREFIX}total" + ] + + +def test_cache_miss_without_pre_call_still_emits_no_span(): + """The cache-hit relaxation is not a general loosening: a success callback + with no ``pre_call`` carrier and no cache hit still emits nothing.""" + logger, exporter = _logger() + payload = _payload(cache_hit=False, response_cost=0.002) + asyncio.run( + logger.async_log_success_event(_kwargs(payload=payload), None, None, None) + ) + assert exporter.get_finished_spans() == () + + def test_real_llm_failure_still_emitted(): """A genuine LLM failure: ``pre_call`` ran (the call was attempted), so the CLIENT span is opened at the boundary and closed ERROR.""" From 1051c9c1008ede63a1b0bb40bb81575430866b8c Mon Sep 17 00:00:00 2001 From: jesus Date: Mon, 28 Sep 2026 22:43:07 +0000 Subject: [PATCH 2/3] fix(otel): read cache payload fields defensively without casts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/integrations/otel/model/metadata.py | 2 +- litellm/integrations/otel/model/payloads.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/litellm/integrations/otel/model/metadata.py b/litellm/integrations/otel/model/metadata.py index 6f2699335a0..6a1b2686273 100644 --- a/litellm/integrations/otel/model/metadata.py +++ b/litellm/integrations/otel/model/metadata.py @@ -246,7 +246,7 @@ class LLMCallEvent: auth_metadata=auth_metadata(payload, kwargs), is_no_upstream_call=bool(kwargs.get(LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL)), upstream_started=kwargs.get("api_call_start_time") is not None, - cache_hit=bool(cast("Mapping[str, object]", payload).get("cache_hit")) if payload else False, + cache_hit=bool(payload.get("cache_hit")) if payload else False, # pyright: ignore[reportUnknownMemberType] # payload dicts may omit this key provisional_span_name=f"{operation.value} {model}".strip(), time_to_first_chunk_seconds=time_to_first_chunk_seconds(kwargs), trace=trace, diff --git a/litellm/integrations/otel/model/payloads.py b/litellm/integrations/otel/model/payloads.py index a41b0eb2562..e0dea7d0b4a 100644 --- a/litellm/integrations/otel/model/payloads.py +++ b/litellm/integrations/otel/model/payloads.py @@ -471,8 +471,8 @@ class LLMCallSpanData: trace=trace or TraceControls(), session_id=session_id or None, embedding_output=embedding_output if capture_content else None, - cache_hit=as_bool(cast("Mapping[str, object]", payload).get("cache_hit")), - saved_cache_cost=as_float(cast("Mapping[str, object]", payload).get("saved_cache_cost")), + cache_hit=as_bool(payload.get("cache_hit")), # pyright: ignore[reportUnknownMemberType] # payload dicts may omit this key + saved_cache_cost=as_float(payload.get("saved_cache_cost")), # pyright: ignore[reportUnknownMemberType] # payload dicts may omit this key ) From 444db7c1e8ca4befcbc910c869d5406f5255d0a8 Mon Sep 17 00:00:00 2001 From: jesus Date: Mon, 28 Sep 2026 22:58:40 +0000 Subject: [PATCH 3/3] chore(otel): clarify pyright suppression reason on cache payload reads Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/integrations/otel/model/metadata.py | 2 +- litellm/integrations/otel/model/payloads.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/litellm/integrations/otel/model/metadata.py b/litellm/integrations/otel/model/metadata.py index 6a1b2686273..773cc7043c4 100644 --- a/litellm/integrations/otel/model/metadata.py +++ b/litellm/integrations/otel/model/metadata.py @@ -246,7 +246,7 @@ class LLMCallEvent: auth_metadata=auth_metadata(payload, kwargs), is_no_upstream_call=bool(kwargs.get(LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL)), upstream_started=kwargs.get("api_call_start_time") is not None, - cache_hit=bool(payload.get("cache_hit")) if payload else False, # pyright: ignore[reportUnknownMemberType] # payload dicts may omit this key + cache_hit=bool(payload.get("cache_hit")) if payload else False, # pyright: ignore[reportUnknownMemberType] # TypedDict .get on a str key is partially unknown provisional_span_name=f"{operation.value} {model}".strip(), time_to_first_chunk_seconds=time_to_first_chunk_seconds(kwargs), trace=trace, diff --git a/litellm/integrations/otel/model/payloads.py b/litellm/integrations/otel/model/payloads.py index e0dea7d0b4a..aff7b56e844 100644 --- a/litellm/integrations/otel/model/payloads.py +++ b/litellm/integrations/otel/model/payloads.py @@ -471,8 +471,8 @@ class LLMCallSpanData: trace=trace or TraceControls(), session_id=session_id or None, embedding_output=embedding_output if capture_content else None, - cache_hit=as_bool(payload.get("cache_hit")), # pyright: ignore[reportUnknownMemberType] # payload dicts may omit this key - saved_cache_cost=as_float(payload.get("saved_cache_cost")), # pyright: ignore[reportUnknownMemberType] # payload dicts may omit this key + cache_hit=as_bool(payload.get("cache_hit")), # pyright: ignore[reportUnknownMemberType] # TypedDict .get on a str key is partially unknown + saved_cache_cost=as_float(payload.get("saved_cache_cost")), # pyright: ignore[reportUnknownMemberType] # TypedDict .get on a str key is partially unknown )