fix(messages): translate GenericResponseOutputItem output instead of dropping it

translate_response() dispatches output items against the three openai-SDK
types and dict. GenericResponseOutputItem -- litellm's own wrapper, built
by the chat-completions->Responses bridge and the proxy MCP handler, and
by the model_construct fallback for provider dialects that fail
model_validate -- matches none of them, so every such item was silently
skipped: /v1/messages answered content: [] with stop_reason "end_turn"
and usage intact.

Normalise non-SDK, non-dict items via model_dump() so the existing dict
branch handles them; SDK types and plain dicts keep their branches.
This commit is contained in:
Mike Hummel 2026-09-07 16:57:34 +02:00
parent 1af7a403c6
commit 0c27e729e7
2 changed files with 137 additions and 1 deletions

View file

@ -632,7 +632,25 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
cast(Iterable[object], response.output) # cast-ok: output items re-validated per item
)
for item in response.output:
# An upstream whose dialect fails ResponsesAPIResponse.model_validate
# reaches this function through the model_construct fallback, whose
# output items are GenericResponseOutputItem -- litellm's own wrapper,
# not an openai-SDK type and not a dict. Without normalisation none of
# the branches below matches and every item is silently skipped.
_sdk_item_types: Final = (ResponseReasoningItem, ResponseOutputMessage, ResponseFunctionToolCall)
def _normalised(item: object) -> object:
if isinstance(item, (dict, *_sdk_item_types)):
return item
_dump = getattr(item, "model_dump", None)
if callable(_dump):
_dumped = _dump()
return _dumped if isinstance(_dumped, dict) else item
return item
output_items: Final = tuple(_normalised(_raw_item) for _raw_item in response.output)
for item in output_items:
if isinstance(item, ResponseReasoningItem):
content.extend(self._thinking_blocks_from_reasoning_item(item.summary))

View file

@ -1508,6 +1508,124 @@ class TestTranslateResponse:
assert result["stop_reason"] == "tool_use"
class TestTranslateResponseGenericOutputItems:
"""GenericResponseOutputItem output must not be silently dropped.
litellm itself manufactures GenericResponseOutputItem objects for
ResponsesAPIResponse.output (the chat-completions bridge in
responses/litellm_completion_transformation/transformation.py and the
proxy MCP handler), and on the model_construct fallback path of the
openai responses transformation (observed on v1.100.0 with a gateway
whose reasoning items carry content[].output_text instead of summary,
which fails model_validate). GenericResponseOutputItem is not an
openai-SDK type and not a dict, so the isinstance dispatch in
translate_response matched none of its branches and dropped every
item: /v1/messages answered with an empty content array while usage
flowed through.
"""
@staticmethod
def _make_generic_item(item_type: str, **overrides: Any) -> Any:
"""Build a GenericResponseOutputItem the way the completion bridge does."""
from litellm.types.responses.main import GenericResponseOutputItem, OutputText
base: Dict[str, Any] = {
"type": item_type,
"id": "item_1",
"status": "completed",
"role": "assistant",
"content": [OutputText(type="output_text", text="hi", annotations=[])],
}
base.update(overrides)
return GenericResponseOutputItem(**base)
def test_generic_message_item_becomes_text_block(self):
response = _make_mock_response(
output=[
self._make_generic_item(
"reasoning",
id="rs_1",
content=[
{
"type": "output_text",
"text": "thinking about the command",
"annotations": [],
}
],
),
self._make_generic_item(
"message",
id="msg_1",
content=[
{
"type": "output_text",
"text": "<verdict>safe</verdict>",
"annotations": [],
}
],
),
]
)
result: Any = _ADAPTER.translate_response(response)
text_blocks = [b for b in result["content"] if b.get("type") == "text"]
assert len(text_blocks) == 1
assert text_blocks[0]["text"] == "<verdict>safe</verdict>"
assert result["stop_reason"] == "end_turn"
def test_generic_function_call_item_becomes_tool_use(self):
fc = self._make_generic_item(
"function_call",
call_id="call_1",
name="get_weather",
arguments='{"city": "NYC"}',
content=[],
)
response = _make_mock_response(output=[fc])
result: Any = _ADAPTER.translate_response(response)
assert len(result["content"]) == 1
block = result["content"][0]
assert block["type"] == "tool_use"
assert block["id"] == "call_1"
assert block["name"] == "get_weather"
assert block["input"] == {"city": "NYC"}
assert result["stop_reason"] == "tool_use"
def test_model_construct_output_survives(self):
"""End-to-end fallback shape: whatever pydantic's model_construct
yields for a raw provider payload (dict or GenericResponseOutputItem,
depending on the union's first arm), the content must arrive."""
from litellm.types.llms.openai import ResponsesAPIResponse
response = ResponsesAPIResponse.model_construct(
id="resp_generic_1",
created_at=1788788375,
model="glm-5.3-flash",
status="completed",
output=[
{
"type": "message",
"id": "msg_1",
"status": "completed",
"role": "assistant",
"content": [
{
"type": "output_text",
"text": "<verdict>safe</verdict>",
"annotations": [],
}
],
}
],
)
result: Any = _ADAPTER.translate_response(response)
text_blocks = [b for b in result["content"] if b.get("type") == "text"]
assert len(text_blocks) == 1
assert text_blocks[0]["text"] == "<verdict>safe</verdict>"
class TestToolResultImages:
"""Images inside tool_result blocks must survive translation: the
function_call_output carries a text placeholder and the image is sent as an