fix(logging): translate Responses bridge result to ModelResponse for spend logs (#28985)

PR #29394 fixed the AnthropicResponse.model_validate crash for the streaming
anthropic_messages -> OpenAI Responses bridge by unwrapping terminal events
and returning the inner ResponsesAPIResponse. The spend_logs row lands and
usage/cost are correct, but the row's response field stores the Responses
API shape (output[...].content[...].text). The proxy UI Logs tab reads
response.choices[0].message via parseMessages in prettyMessagesUtils.ts
with no fallback for the Responses shape, so the OutputCard renders "No
response data available" for every cross-routed call. The same shape
mismatch affects every downstream consumer of spend_logs that assumes the
canonical chat-completion shape

This change keeps the unwrap from #29394 but routes the resulting
ResponsesAPIResponse (and the bare-response non-streaming path) through
LiteLLMResponsesTransformationHandler.transform_response, which is the
same conversion already used by the chat-completion Responses bridge.
Spend_logs now stores a ModelResponse with choices[0].message.content, so
the UI and other consumers see the assistant text. On a translation
failure (eg. empty output on an incomplete response) the handler falls
back to a minimal ModelResponse carrying model and usage so the row still
lands rather than being dropped as a Non-Blocking error

Also corrects a stale comment in the Responses adapter that implied the
call type was reclassified to acompletion; the code preserves
anthropic_messages and the success handler translates back to
ModelResponse for the row

Fixes #28595
This commit is contained in:
Piotr Minkina 2026-06-11 13:57:05 +02:00 • committed by GitHub
parent c5df3f2260
commit 706ea8f3d6
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 147 additions and 32 deletions

View file

@ -3508,9 +3508,7 @@ class Logging(LiteLLMLoggingBaseClass):
else:
return None
def _handle_anthropic_messages_response_logging(
self, result: Any
) -> Union[ModelResponse, ResponsesAPIResponse]:
def _handle_anthropic_messages_response_logging(self, result: Any) -> ModelResponse:
"""
Handles logging for Anthropic messages responses.
@ -3529,15 +3527,14 @@ class Logging(LiteLLMLoggingBaseClass):
return result
elif isinstance(result, ModelResponse):
return result
elif isinstance(
if isinstance(
result,
(ResponseCompletedEvent, ResponseIncompleteEvent, ResponseFailedEvent),
):
# anthropic_messages() can route to OpenAI Responses API; in that path
# the assembled streaming result is one of these terminal events rather than
# a ModelResponse. Return the inner response so downstream handlers
# (_transform_usage_objects, normalize_logging_result) can process it.
return result.response
result = result.response
if isinstance(result, ResponsesAPIResponse):
return self._translate_responses_api_response_to_model_response(result)
httpx_response = self.model_call_details.get("httpx_response", None)
if httpx_response and isinstance(httpx_response, httpx.Response):
@ -3571,6 +3568,55 @@ class Logging(LiteLLMLoggingBaseClass):
)
return result
def _translate_responses_api_response_to_model_response(
self, result: ResponsesAPIResponse
) -> ModelResponse:
"""
Convert a Responses API response into a ModelResponse for spend_logs.
The proxy UI parses spend_log rows expecting chat-completion shape
(response.choices[0].message); a raw ResponsesAPIResponse dump (output[...])
would render as empty in the Logs tab. Translation also yields full
choices/message detail downstream consumers can rely on.
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
)
try:
return LiteLLMResponsesTransformationHandler().transform_response(
model=self.model,
raw_response=result,
model_response=litellm.ModelResponse(),
logging_obj=self,
request_data={},
messages=[],
optional_params={},
litellm_params={},
encoding=litellm.encoding,
)
except Exception as e:
verbose_logger.debug(
"Responses API -> ModelResponse translation failed for "
"anthropic_messages logging (%s); falling back to minimal "
"usage-only ModelResponse to keep the spend_logs row.",
str(e),
)
model_response = litellm.ModelResponse()
model_response.model = self.model
usage = getattr(result, "usage", None)
if usage is not None and ResponseAPILoggingUtils._is_response_api_usage(
usage
):
setattr(
model_response,
"usage",
ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
usage
),
)
return model_response
def _handle_non_streaming_google_genai_generate_content_response_logging(
self, result: Any
) -> ModelResponse:

View file

@ -102,9 +102,9 @@ def _build_responses_kwargs(
from litellm.types.utils import CallTypes
if isinstance(value, LiteLLMLoggingObject):
# Reclassify as acompletion so the success handler doesn't try to
# validate the Responses API event as an AnthropicResponse.
# (Mirrors the pattern used in LiteLLMMessagesToCompletionTransformationHandler.)
# Keep call_type as anthropic_messages so spend_logs are billed
# against /v1/messages; the success handler translates the
# Responses API result back to a ModelResponse for the row.
setattr(value, "call_type", CallTypes.anthropic_messages.value)
responses_kwargs[key] = value
elif key not in excluded and key not in responses_kwargs and value is not None:

View file

@ -3151,6 +3151,41 @@ def test_get_error_information_prefers_message_attribute_over_empty_str():
assert info["error_code"] == "401"
def _anthropic_messages_logging_obj():
return LitellmLogging(
model="openai/my-local",
messages=[{"role": "user", "content": "hi"}],
stream=True,
call_type="anthropic_messages",
start_time=time.time(),
litellm_call_id="28595",
function_id="28595",
)
def _responses_api_response_with_text(text="hello world"):
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
return ResponsesAPIResponse(
id="resp-28595",
created_at=1700000000,
output=[
ResponseOutputMessage(
id="msg-1",
type="message",
role="assistant",
status="completed",
content=[
ResponseOutputText(annotations=[], text=text, type="output_text")
],
)
],
usage=ResponseAPIUsage(input_tokens=11, output_tokens=7, total_tokens=18),
)
@pytest.mark.parametrize(
"event_cls, event_type",
[
@ -3159,34 +3194,68 @@ def test_get_error_information_prefers_message_attribute_over_empty_str():
("ResponseFailedEvent", "response.failed"),
],
)
def test_handle_anthropic_messages_response_logging_with_terminal_responses_api_events(
def test_handle_anthropic_messages_response_logging_translates_terminal_responses_api_event(
event_cls, event_type
):
"""Regression test for #28943: when anthropic_messages routes to OpenAI Responses
API and stream=True, success_handler receives a terminal ResponsesAPI event instead
of a ModelResponse. The handler must return the inner ResponsesAPIResponse rather
than crashing with AnthropicResponse.model_validate."""
"""Regression for #28595 / #28943. When anthropic_messages routes to the OpenAI
Responses backend and stream=True, success_handler receives a terminal Responses
API event. The handler must translate it to a ModelResponse whose choices carry
the assistant text, so the proxy UI Logs tab (which reads response.choices[0])
renders the response content instead of "No response data available"."""
import importlib
openai_types = importlib.import_module("litellm.types.llms.openai")
EventClass = getattr(openai_types, event_cls)
from litellm.types.llms.openai import ResponsesAPIResponse
logging_obj = LitellmLogging(
model="gpt-4o",
messages=[{"role": "user", "content": "hello"}],
stream=True,
call_type="anthropic_messages",
start_time=time.time(),
litellm_call_id="test-rce-123",
function_id="test-fn",
)
inner_response = ResponsesAPIResponse(
id="resp_test", created_at=1700000000, output=[]
)
logging_obj = _anthropic_messages_logging_obj()
inner_response = _responses_api_response_with_text("hello world")
event = EventClass(type=event_type, response=inner_response)
result = logging_obj._handle_anthropic_messages_response_logging(result=event)
assert result is inner_response
assert isinstance(result, ModelResponse)
assert result.choices[0].message.content == "hello world" # type: ignore[union-attr]
assert result.usage.prompt_tokens == 11 # type: ignore[attr-defined]
assert result.usage.completion_tokens == 7 # type: ignore[attr-defined]
def test_handle_anthropic_messages_response_logging_translates_bare_responses_api_response():
"""Non-streaming bridge path: result is a bare ResponsesAPIResponse (no event wrap)."""
logging_obj = _anthropic_messages_logging_obj()
result = logging_obj._handle_anthropic_messages_response_logging(
result=_responses_api_response_with_text("hi there")
)
assert isinstance(result, ModelResponse)
assert result.choices[0].message.content == "hi there" # type: ignore[union-attr]
assert result.usage.total_tokens == 18 # type: ignore[attr-defined]
def test_handle_anthropic_messages_response_logging_passes_model_response_through():
"""Anthropic-native path already yields a ModelResponse; it must be returned unchanged."""
logging_obj = _anthropic_messages_logging_obj()
model_response = ModelResponse()
assert (
logging_obj._handle_anthropic_messages_response_logging(result=model_response)
is model_response
)
def test_handle_anthropic_messages_response_logging_degrades_on_unparseable_responses_payload():
"""If the Responses translation raises (eg. empty output on an incomplete response),
the row must still land: a minimal ModelResponse with model + usage is returned."""
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
logging_obj = _anthropic_messages_logging_obj()
empty = ResponsesAPIResponse(
id="resp-empty",
created_at=1700000000,
output=[],
usage=ResponseAPIUsage(input_tokens=4, output_tokens=0, total_tokens=4),
)
result = logging_obj._handle_anthropic_messages_response_logging(result=empty)
assert isinstance(result, ModelResponse)
assert result.model == "openai/my-local"
assert result.usage.prompt_tokens == 4 # type: ignore[attr-defined]