fix(responses_adapter): read cache_read_tokens from input_tokens_details on Responses usage

The Anthropic→Responses streaming adapter was reading
usage.cache_read_input_tokens (an Anthropic-only field) and so always
emitted cache_read_input_tokens=0 to Anthropic-format clients, even
when OpenAI's underlying response correctly reported a non-zero
cached_tokens. This broke observability for Anthropic-compatible
clients (Claude Code, Claude Agent SDK, dashboards, Sentry, Langfuse,
billing) routing OpenAI Responses traffic through LiteLLM.

Read cached prompt tokens from usage.input_tokens_details.cached_tokens
(parallel to the existing Chat Completions path's use of
prompt_tokens_details.cached_tokens), fall back to the Anthropic-named
field if the upstream provider already speaks Anthropic usage, and
subtract cached from input_tokens to preserve Anthropic's 'uncached'
semantics for the input_tokens count.

Also drops the dead first assignment that stored
input_tokens_details / output_tokens_details into the cache token
variables before immediately overwriting them.

Fixes #28354
This commit is contained in:
Tai An 2026-05-20 12:04:49 -07:00
parent e182a5e0ba
commit b31acebb7d

View file

@ -257,17 +257,33 @@ class AnthropicResponsesStreamWrapper:
stop_reason = "max_tokens"
usage = getattr(response_obj, "usage", None)
if usage is not None:
input_tokens = getattr(usage, "input_tokens", 0) or 0
output_tokens = getattr(usage, "output_tokens", 0) or 0
cache_creation_tokens = getattr(usage, "input_tokens_details", None) # type: ignore[assignment]
cache_read_tokens = getattr(usage, "output_tokens_details", None) # type: ignore[assignment]
# Prefer direct cache fields if present
input_tokens = int(getattr(usage, "input_tokens", 0) or 0)
output_tokens = int(getattr(usage, "output_tokens", 0) or 0)
# OpenAI Responses API exposes cached prompt tokens at
# ``usage.input_tokens_details.cached_tokens`` (parallel to
# ``usage.prompt_tokens_details.cached_tokens`` on Chat Completions).
itd = getattr(usage, "input_tokens_details", None)
if itd is not None:
if isinstance(itd, dict):
cache_read_tokens = int(itd.get("cached_tokens", 0) or 0)
else:
cache_read_tokens = int(
getattr(itd, "cached_tokens", 0) or 0
)
# Fall back to Anthropic-style direct fields if the upstream
# provider already speaks Anthropic usage.
if not cache_read_tokens:
cache_read_tokens = int(
getattr(usage, "cache_read_input_tokens", 0) or 0
)
cache_creation_tokens = int(
getattr(usage, "cache_creation_input_tokens", 0) or 0
)
cache_read_tokens = int(
getattr(usage, "cache_read_input_tokens", 0) or 0
)
# Anthropic semantics: ``input_tokens`` is the *uncached* prompt
# count. OpenAI reports total prompt tokens (cached + uncached),
# so subtract to keep client-side accounting consistent.
if cache_read_tokens and input_tokens >= cache_read_tokens:
input_tokens = input_tokens - cache_read_tokens
# Check if tool_use was in the output to override stop_reason
if response_obj is not None: