mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(anthropic): stop emitting empty thinking blocks on the Responses adapter
OpenAI emits a reasoning output item on every reasoning turn, but only emits
reasoning_summary_text deltas when a summary was requested and actually
produced. The Anthropic /v1/messages Responses stream adapter opened the
thinking content block eagerly on response.output_item.added, so a summary-less
reasoning item surfaced as {"type": "thinking", "thinking": ""}. Clients persist
that in their session transcript and replay it on the next turn; an Anthropic
model then rejects the request with "each thinking block must contain thinking",
which is what users hit when a resumed session falls back to the default
Anthropic model.
Open the thinking block on the first non-empty summary delta instead, and only
emit content_block_stop for items that actually have an open block.
This commit is contained in:
parent
ae53de36e8
commit
ebf6167d8a
2 changed files with 106 additions and 56 deletions
|
|
@ -68,6 +68,19 @@ class AnthropicResponsesStreamWrapper:
|
|||
self._current_block_index += 1
|
||||
return self._current_block_index
|
||||
|
||||
def _open_block(self, item_id: str | None, content_block: dict[str, Any]) -> int:
|
||||
block_idx = self._next_block_index()
|
||||
if item_id:
|
||||
self._item_id_to_block_index[item_id] = block_idx
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_start",
|
||||
"index": block_idx,
|
||||
"content_block": content_block,
|
||||
}
|
||||
)
|
||||
return block_idx
|
||||
|
||||
def _process_event(self, event: Any) -> None:
|
||||
"""Convert one Responses API event into zero or more Anthropic chunks queued for emission."""
|
||||
event_type = getattr(event, "type", None)
|
||||
|
|
@ -93,47 +106,22 @@ class AnthropicResponsesStreamWrapper:
|
|||
item_id = getattr(item, "id", None) or (item.get("id") if isinstance(item, dict) else None)
|
||||
|
||||
if item_type == "message":
|
||||
block_idx = self._next_block_index()
|
||||
if item_id:
|
||||
self._item_id_to_block_index[item_id] = block_idx
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_start",
|
||||
"index": block_idx,
|
||||
"content_block": {"type": "text", "text": ""},
|
||||
}
|
||||
)
|
||||
self._open_block(item_id, {"type": "text", "text": ""})
|
||||
elif item_type == "function_call":
|
||||
call_id: Final = (
|
||||
getattr(item, "call_id", None) or (item.get("call_id") if isinstance(item, dict) else None) or ""
|
||||
)
|
||||
name = getattr(item, "name", None) or (item.get("name") if isinstance(item, dict) else None) or ""
|
||||
block_idx = self._next_block_index()
|
||||
if item_id:
|
||||
self._item_id_to_block_index[item_id] = block_idx
|
||||
self._pending_tool_ids[item_id] = call_id
|
||||
self._chunk_queue.append(
|
||||
self._open_block(
|
||||
item_id,
|
||||
{
|
||||
"type": "content_block_start",
|
||||
"index": block_idx,
|
||||
"content_block": {
|
||||
"type": "tool_use",
|
||||
"id": call_id,
|
||||
"name": name,
|
||||
"input": {},
|
||||
},
|
||||
}
|
||||
)
|
||||
elif item_type == "reasoning":
|
||||
block_idx = self._next_block_index()
|
||||
if item_id:
|
||||
self._item_id_to_block_index[item_id] = block_idx
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_start",
|
||||
"index": block_idx,
|
||||
"content_block": {"type": "thinking", "thinking": ""},
|
||||
}
|
||||
"type": "tool_use",
|
||||
"id": call_id,
|
||||
"name": name,
|
||||
"input": {},
|
||||
},
|
||||
)
|
||||
return
|
||||
|
||||
|
|
@ -146,16 +134,7 @@ class AnthropicResponsesStreamWrapper:
|
|||
# Some providers (e.g. LMStudio) skip response.output_item.added,
|
||||
# so no text block is open yet; synthesize content_block_start
|
||||
# instead of emitting a delta with index -1
|
||||
block_idx = self._next_block_index()
|
||||
if item_id:
|
||||
self._item_id_to_block_index[item_id] = block_idx
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_start",
|
||||
"index": block_idx,
|
||||
"content_block": {"type": "text", "text": ""},
|
||||
}
|
||||
)
|
||||
block_idx = self._open_block(item_id, {"type": "text", "text": ""})
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_delta",
|
||||
|
|
@ -169,11 +148,11 @@ class AnthropicResponsesStreamWrapper:
|
|||
if event_type == "response.reasoning_summary_text.delta":
|
||||
item_id = getattr(event, "item_id", None) or (event.get("item_id") if isinstance(event, dict) else None)
|
||||
delta = getattr(event, "delta", "") or (event.get("delta", "") if isinstance(event, dict) else "")
|
||||
block_idx = (
|
||||
self._item_id_to_block_index.get(item_id, self._current_block_index)
|
||||
if item_id
|
||||
else self._current_block_index
|
||||
)
|
||||
block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index
|
||||
if block_idx < 0:
|
||||
if not delta:
|
||||
return
|
||||
block_idx = self._open_block(item_id, {"type": "thinking", "thinking": ""})
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_delta",
|
||||
|
|
@ -207,11 +186,9 @@ class AnthropicResponsesStreamWrapper:
|
|||
item_id = (
|
||||
getattr(item, "id", None) or (item.get("id") if isinstance(item, dict) else None) if item else None
|
||||
)
|
||||
block_idx = (
|
||||
self._item_id_to_block_index.get(item_id, self._current_block_index)
|
||||
if item_id
|
||||
else self._current_block_index
|
||||
)
|
||||
block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index
|
||||
if block_idx < 0:
|
||||
return
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_stop",
|
||||
|
|
|
|||
|
|
@ -76,6 +76,78 @@ class TestProcessEventResponseCreatedGuard:
|
|||
assert len(message_starts) == 1
|
||||
|
||||
|
||||
class TestReasoningItemWithoutSummaryText:
|
||||
"""Regression: a reasoning item whose summary never produces text must not
|
||||
surface as a thinking content block.
|
||||
|
||||
OpenAI emits ``response.output_item.added`` with ``type: "reasoning"`` on
|
||||
every reasoning turn, but only emits
|
||||
``response.reasoning_summary_text.delta`` when a summary was requested and
|
||||
the model actually produced one. Eagerly opening the block on
|
||||
``output_item.added`` left ``{"type": "thinking", "thinking": ""}`` in the
|
||||
assistant turn, which clients persist in their session transcript. Replaying
|
||||
that transcript against an Anthropic model (what ``claude --resume`` does
|
||||
once the resumed session falls back to the default Anthropic model) fails
|
||||
with::
|
||||
|
||||
400 invalid_request_error - messages.2.content.0.thinking:
|
||||
each thinking block must contain thinking
|
||||
|
||||
So the thinking block is opened on the first non-empty summary delta.
|
||||
"""
|
||||
|
||||
@staticmethod
|
||||
def _gpt_turn(reasoning_summary_deltas: list) -> list:
|
||||
return [
|
||||
{"type": "response.created"},
|
||||
{"type": "response.output_item.added", "item": {"type": "reasoning", "id": "rs_1"}},
|
||||
*(
|
||||
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": delta}
|
||||
for delta in reasoning_summary_deltas
|
||||
),
|
||||
{"type": "response.output_item.done", "item": {"type": "reasoning", "id": "rs_1"}},
|
||||
{"type": "response.output_item.added", "item": {"type": "message", "id": "msg_1"}},
|
||||
{"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Hello"},
|
||||
{"type": "response.output_item.done", "item": {"type": "message", "id": "msg_1"}},
|
||||
]
|
||||
|
||||
def test_reasoning_without_summary_emits_no_thinking_block(self):
|
||||
chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=[]))
|
||||
|
||||
assert not [
|
||||
c for c in chunks if c["type"] == "content_block_start" and c["content_block"]["type"] == "thinking"
|
||||
]
|
||||
assert [(c["type"], c.get("index")) for c in chunks[1:]] == [
|
||||
("content_block_start", 0),
|
||||
("content_block_delta", 0),
|
||||
("content_block_stop", 0),
|
||||
]
|
||||
assert chunks[1]["content_block"] == {"type": "text", "text": ""}
|
||||
|
||||
def test_reasoning_with_only_empty_summary_deltas_emits_no_thinking_block(self):
|
||||
chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=["", ""]))
|
||||
|
||||
assert not [c for c in chunks if c["type"] == "content_block_delta" and c["delta"]["type"] == "thinking_delta"]
|
||||
assert not [
|
||||
c for c in chunks if c["type"] == "content_block_start" and c["content_block"]["type"] == "thinking"
|
||||
]
|
||||
|
||||
def test_reasoning_with_summary_text_still_emits_a_thinking_block(self):
|
||||
chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=["Weigh", "ing options"]))
|
||||
|
||||
assert [(c["type"], c.get("index")) for c in chunks[1:]] == [
|
||||
("content_block_start", 0),
|
||||
("content_block_delta", 0),
|
||||
("content_block_delta", 0),
|
||||
("content_block_stop", 0),
|
||||
("content_block_start", 1),
|
||||
("content_block_delta", 1),
|
||||
("content_block_stop", 1),
|
||||
]
|
||||
assert chunks[1]["content_block"] == {"type": "thinking", "thinking": ""}
|
||||
assert "".join(c["delta"]["thinking"] for c in chunks[2:4]) == "Weighing options"
|
||||
|
||||
|
||||
class TestProcessEventTextDeltaWithoutOutputItemAdded:
|
||||
"""Streams that skip response.output_item.added (e.g. LMStudio) must still
|
||||
open a text block before any delta and never emit index -1."""
|
||||
|
|
@ -110,12 +182,13 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded:
|
|||
"type": "response.output_item.added",
|
||||
"item": {"type": "reasoning", "id": "rs_1"},
|
||||
},
|
||||
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "hm"},
|
||||
{"type": "response.output_text.delta", "item_id": "m1", "delta": "Hi"},
|
||||
]
|
||||
)
|
||||
assert chunks[1]["type"] == "content_block_start"
|
||||
assert chunks[1]["content_block"] == {"type": "text", "text": ""}
|
||||
assert [c["index"] for c in chunks[1:]] == [1, 1]
|
||||
assert chunks[2]["type"] == "content_block_start"
|
||||
assert chunks[2]["content_block"] == {"type": "text", "text": ""}
|
||||
assert [c["index"] for c in chunks[2:]] == [1, 1]
|
||||
|
||||
def test_process_event_registered_item_id_does_not_synthesize_start(self):
|
||||
chunks = _process_all(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue