mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(otel): emit V2 LLM-call span for cache hits with explicit cache attributes
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
7e383c9f6a
commit
0a8ae7e624
7 changed files with 66 additions and 2 deletions
|
|
@ -508,8 +508,12 @@ class OpenTelemetryV2(CustomLogger):
|
|||
# logger is a success/failure callback only, so ``pre_call`` never reaches it
|
||||
# and no carrier exists. The payload plus the request-level provider-handoff
|
||||
# stamp (``upstream_started``) is the affirmative signal of a real call; a
|
||||
# gate rejection carries ``is_no_upstream_call`` and gets no span.
|
||||
if carrier is None and (call.is_no_upstream_call or not call.upstream_started or call.payload is None):
|
||||
# gate rejection carries ``is_no_upstream_call`` and gets no span. A cache
|
||||
# hit never hands off to a provider, so ``cache_hit`` substitutes for the
|
||||
# upstream stamp — the payload proves the call completed.
|
||||
if carrier is None and (
|
||||
call.is_no_upstream_call or call.payload is None or (not call.upstream_started and not call.cache_hit)
|
||||
):
|
||||
return None
|
||||
try:
|
||||
return self._finish_carrier(carrier, call, start_time, end_time)
|
||||
|
|
|
|||
|
|
@ -89,6 +89,8 @@ class GenAIMapper:
|
|||
f"{LiteLLM.COST_PREFIX}margin_fixed_amount": lambda d: d.cost.margin_fixed_amount,
|
||||
f"{LiteLLM.COST_PREFIX}margin_percent": lambda d: d.cost.margin_percent,
|
||||
f"{LiteLLM.COST_PREFIX}margin_total_amount": lambda d: d.cost.margin_total_amount,
|
||||
LiteLLM.CACHE_HIT: lambda d: d.cache_hit,
|
||||
LiteLLM.SAVED_CACHE_COST: lambda d: d.saved_cache_cost,
|
||||
LiteLLM.REQUEST_STREAMING: lambda d: d.is_streaming,
|
||||
LiteLLM.REQUEST_ROUTE: lambda d: d.request_route,
|
||||
}
|
||||
|
|
|
|||
|
|
@ -220,6 +220,10 @@ class LLMCallEvent:
|
|||
# actually attempted — router pre-call rejections, SDK failures before the
|
||||
# provider handoff, and standalone guardrail runs all lack it.
|
||||
upstream_started: bool
|
||||
# True for a response served from the litellm response cache. A cache hit
|
||||
# never hands off to a provider, so ``upstream_started`` stays False even
|
||||
# though the call completed and is worth a span.
|
||||
cache_hit: bool
|
||||
# A best-effort ``"{operation} {model}"`` name known at ``pre_call`` time. The
|
||||
# span is renamed from the typed payload at close (``finish_span``); this only
|
||||
# needs to be reasonable for a span that never gets closed (a leak).
|
||||
|
|
@ -242,6 +246,7 @@ class LLMCallEvent:
|
|||
auth_metadata=auth_metadata(payload, kwargs),
|
||||
is_no_upstream_call=bool(kwargs.get(LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL)),
|
||||
upstream_started=kwargs.get("api_call_start_time") is not None,
|
||||
cache_hit=bool(cast("Mapping[str, object]", payload).get("cache_hit")) if payload else False,
|
||||
provisional_span_name=f"{operation.value} {model}".strip(),
|
||||
time_to_first_chunk_seconds=time_to_first_chunk_seconds(kwargs),
|
||||
trace=trace,
|
||||
|
|
|
|||
|
|
@ -411,6 +411,8 @@ class LLMCallSpanData:
|
|||
trace: TraceControls = field(default_factory=TraceControls)
|
||||
session_id: str | None = None
|
||||
embedding_output: EmbeddingOutput | None = None
|
||||
cache_hit: bool | None = None
|
||||
saved_cache_cost: float | None = None
|
||||
|
||||
@classmethod
|
||||
def from_standard_logging_payload(
|
||||
|
|
@ -469,6 +471,8 @@ class LLMCallSpanData:
|
|||
trace=trace or TraceControls(),
|
||||
session_id=session_id or None,
|
||||
embedding_output=embedding_output if capture_content else None,
|
||||
cache_hit=as_bool(cast("Mapping[str, object]", payload).get("cache_hit")),
|
||||
saved_cache_cost=as_float(cast("Mapping[str, object]", payload).get("saved_cache_cost")),
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -332,6 +332,10 @@ class LiteLLM:
|
|||
# semconv key for an MCP server's *name* (the convention uses ``server.address``
|
||||
# for its network location), so it lives under the vendor namespace.
|
||||
MCP_SERVER_NAME: Final = "litellm.mcp.server.name"
|
||||
# Whether the response was served from the litellm response cache instead of
|
||||
# an upstream provider call, and the USD cost the hit avoided.
|
||||
CACHE_HIT: Final = "litellm.cache_hit"
|
||||
SAVED_CACHE_COST: Final = "litellm.saved_cache_cost"
|
||||
|
||||
|
||||
class Metric:
|
||||
|
|
|
|||
|
|
@ -355,6 +355,19 @@ def test_genai_mapper_cost_breakdown_absent():
|
|||
assert not any(k.startswith(LiteLLM.COST_PREFIX) and k != f"{LiteLLM.COST_PREFIX}total" for k in attrs)
|
||||
|
||||
|
||||
def test_genai_mapper_cache_hit_attrs():
|
||||
from litellm.integrations.otel.model.semconv import LiteLLM
|
||||
|
||||
attrs = GenAIMapper().map(replace(_full_llm_call(), cache_hit=True, saved_cache_cost=0.002))
|
||||
assert attrs[LiteLLM.CACHE_HIT] is True
|
||||
assert attrs[LiteLLM.SAVED_CACHE_COST] == 0.002
|
||||
|
||||
# The default (None) keeps the span sparse: neither key present.
|
||||
uncached = GenAIMapper().map(_full_llm_call())
|
||||
assert LiteLLM.CACHE_HIT not in uncached
|
||||
assert LiteLLM.SAVED_CACHE_COST not in uncached
|
||||
|
||||
|
||||
def test_llm_cost_from_breakdown_maps_costbreakdown_keys():
|
||||
cost = LLMCost.from_breakdown(
|
||||
{
|
||||
|
|
|
|||
|
|
@ -432,6 +432,38 @@ def test_no_span_when_pre_call_never_ran():
|
|||
assert exporter.get_finished_spans() == () # no phantom LLM span
|
||||
|
||||
|
||||
def test_cache_hit_without_pre_call_emits_zero_cost_span():
|
||||
"""A response served from the litellm cache never hands off to a provider,
|
||||
so ``pre_call`` never runs and ``api_call_start_time`` is never stamped —
|
||||
but the completed call's payload still warrants a span, carrying the cache
|
||||
attributes rather than being dropped like a gate rejection."""
|
||||
logger, exporter = _logger()
|
||||
payload = _payload(cache_hit=True, response_cost=0.0, saved_cache_cost=0.002, cost_breakdown=None)
|
||||
asyncio.run(
|
||||
logger.async_log_success_event(_kwargs(payload=payload), None, None, None)
|
||||
)
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert span.attributes[LiteLLM.CACHE_HIT] is True
|
||||
assert span.attributes[LiteLLM.SAVED_CACHE_COST] == 0.002
|
||||
assert span.attributes[f"{LiteLLM.COST_PREFIX}total"] == 0.0
|
||||
assert not [
|
||||
k
|
||||
for k in span.attributes
|
||||
if k.startswith(LiteLLM.COST_PREFIX) and k != f"{LiteLLM.COST_PREFIX}total"
|
||||
]
|
||||
|
||||
|
||||
def test_cache_miss_without_pre_call_still_emits_no_span():
|
||||
"""The cache-hit relaxation is not a general loosening: a success callback
|
||||
with no ``pre_call`` carrier and no cache hit still emits nothing."""
|
||||
logger, exporter = _logger()
|
||||
payload = _payload(cache_hit=False, response_cost=0.002)
|
||||
asyncio.run(
|
||||
logger.async_log_success_event(_kwargs(payload=payload), None, None, None)
|
||||
)
|
||||
assert exporter.get_finished_spans() == ()
|
||||
|
||||
|
||||
def test_real_llm_failure_still_emitted():
|
||||
"""A genuine LLM failure: ``pre_call`` ran (the call was attempted), so the
|
||||
CLIENT span is opened at the boundary and closed ERROR."""
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue