From d00dfabf6f77d3f152ffd96e1379c71d383bf97f Mon Sep 17 00:00:00 2001 From: Bhavya Shah Date: Sun, 26 Jul 2026 13:39:01 +0530 Subject: [PATCH] fix(anthropic): emit stop_reason tool_use for streamed /v1/messages tool calls ollama_chat streaming reports finish_reason "stop" on the final chunk even when the turn contains a tool call, so the Anthropic Messages adapter emitted stop_reason "end_turn" and clients (Claude Code, the Anthropic SDK) dropped the tool_use block. Track whether a tool_use block was streamed and override the final message_delta to tool_use in both the sync and async paths. Adds regression tests covering the sync and async paths plus a guard that a plain text turn keeps end_turn --- .../adapters/streaming_iterator.py | 32 +++++-- .../test_streaming_iterator_tool_args.py | 86 +++++++++++++++++++ 2 files changed, 110 insertions(+), 8 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py index f02333c34c8..9f0e912d74c 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py @@ -209,6 +209,17 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): type="text", text="", ) + self._emitted_tool_use = False + + def _override_stop_reason_for_tool_use(self, processed_chunk: Any) -> Any: + if ( + self._emitted_tool_use + and isinstance(processed_chunk, dict) + and processed_chunk.get("type") == "message_delta" + and processed_chunk.get("delta", {}).get("stop_reason") == "end_turn" + ): + processed_chunk["delta"]["stop_reason"] = "tool_use" + return processed_chunk def _merge_usage_into_held_stop_reason_chunk(self, chunk: Any) -> Dict[str, Any]: """Merge usage data from ``chunk`` into the held ``message_delta`` chunk. @@ -424,10 +435,12 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): self.holding_stop_reason_chunk is not None and getattr(chunk, "usage", None) is not None ) is_final_chunk = chunk.choices[0].finish_reason is not None - processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic( - response=chunk, - current_content_block_index=self.current_content_block_index, - applied_edits=(self.applied_edits if is_final_chunk and not will_merge_into_held else None), + processed_chunk = self._override_stop_reason_for_tool_use( + LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic( + response=chunk, + current_content_block_index=self.current_content_block_index, + applied_edits=(self.applied_edits if is_final_chunk and not will_merge_into_held else None), + ) ) # Check if this is a usage chunk and we have a held stop_reason chunk @@ -647,10 +660,12 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): self.holding_stop_reason_chunk is not None and getattr(chunk, "usage", None) is not None ) is_final_chunk = chunk.choices[0].finish_reason is not None - processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic( - response=chunk, - current_content_block_index=self.current_content_block_index, - applied_edits=(self.applied_edits if is_final_chunk and not will_merge_into_held else None), + processed_chunk = self._override_stop_reason_for_tool_use( + LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic( + response=chunk, + current_content_block_index=self.current_content_block_index, + applied_edits=(self.applied_edits if is_final_chunk and not will_merge_into_held else None), + ) ) # Check if this is a usage chunk and we have a held stop_reason chunk @@ -901,6 +916,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): # Restore original tool name if it was truncated for OpenAI's 64-char limit if block_type == "tool_use": + self._emitted_tool_use = True # Type narrowing: content_block_start is ToolUseBlock when block_type is "tool_use" from typing import cast diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_streaming_iterator_tool_args.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_streaming_iterator_tool_args.py index bd39e420607..be77770d38d 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_streaming_iterator_tool_args.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_streaming_iterator_tool_args.py @@ -381,3 +381,89 @@ def test_sync_stream_no_extra_delta_when_tool_args_empty(): f"got {len(input_json_deltas)}" ) assert input_json_deltas[0]["delta"]["partial_json"] == '{"location": "NYC"}' + + +def _last_message_delta(events): + deltas = [ + e + for e in events + if isinstance(e, dict) and e.get("type") == "message_delta" + ] + assert deltas, f"expected a message_delta event; got {[e.get('type') for e in events if isinstance(e, dict)]}" + return deltas[-1] + + +def _tool_call_chunk(): + return _make_chunk( + Delta( + content=None, + role="assistant", + tool_calls=[ + ChatCompletionDeltaToolCall( + id="call_abc123", + function=Function(name="get_weather", arguments='{"city": "San Francisco"}'), + type="function", + index=0, + ) + ], + ) + ) + + +def test_sync_stream_stop_reason_is_tool_use_when_finish_reason_stop(): + """ + Regression for #34692: ollama_chat streaming reports finish_reason="stop" + on the final chunk even when the turn contains a tool call. The emitted + message_delta must still carry stop_reason="tool_use" (matching the + non-streaming response and the Anthropic Messages spec), or clients drop + the tool call. + """ + finish_chunk = _make_chunk( + Delta(content=None, role="assistant", tool_calls=None), + finish_reason="stop", + ) + wrapper = AnthropicStreamWrapper( + completion_stream=iter([_tool_call_chunk(), finish_chunk]), + model="test-model", + ) + events = _collect_events_sync(wrapper) + assert _last_message_delta(events)["delta"]["stop_reason"] == "tool_use" + + +@pytest.mark.asyncio +async def test_async_stream_stop_reason_is_tool_use_when_finish_reason_stop(): + """Async counterpart of the #34692 stop_reason regression.""" + finish_chunk = _make_chunk( + Delta(content=None, role="assistant", tool_calls=None), + finish_reason="stop", + ) + + async def mock_stream(): + for c in [_tool_call_chunk(), finish_chunk]: + yield c + + wrapper = AnthropicStreamWrapper( + completion_stream=mock_stream(), + model="test-model", + ) + events = await _collect_events_async(wrapper) + assert _last_message_delta(events)["delta"]["stop_reason"] == "tool_use" + + +def test_sync_stream_stop_reason_stays_end_turn_without_tool_use(): + """ + Guard: a plain text turn ending in finish_reason="stop" must keep + stop_reason="end_turn". The tool_use override must not fire when no + tool_use block was streamed. + """ + text_chunk = _make_chunk(Delta(content="Hello there", role="assistant", tool_calls=None)) + finish_chunk = _make_chunk( + Delta(content=None, role="assistant", tool_calls=None), + finish_reason="stop", + ) + wrapper = AnthropicStreamWrapper( + completion_stream=iter([text_chunk, finish_chunk]), + model="test-model", + ) + events = _collect_events_sync(wrapper) + assert _last_message_delta(events)["delta"]["stop_reason"] == "end_turn"