fix: always use choice index=0 for Anthropic streaming responses (#12666)

- Fixed 'missing finish_reason for choice 1' error with reasoning_effort
- Anthropic sends multiple content blocks with different indices
- OpenAI expects all content in a single choice at index=0
- Added comprehensive tests for text-only, text+tool, and multiple tools
This commit is contained in:
Maksim 2025-07-29 17:42:01 -04:00 • committed by GitHub
parent c44e6017c0
commit f3b1b416d1
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 181 additions and 8 deletions

View file

@ -640,7 +640,8 @@ class ModelResponseIterator:
]
] = None
index = int(chunk.get("index", 0))
# Always use index=0 for OpenAI choice format (fixes multi-choice errors)
index = 0
if type_chunk == "content_block_delta":
"""
Anthropic content chunk

View file

@ -1,12 +1,5 @@
import os
import sys
from unittest.mock import MagicMock
sys.path.insert(
0, os.path.abspath("../../../../..")
) # Adds the parent directory to the system path
from litellm.llms.anthropic.chat.handler import ModelResponseIterator
from litellm.types.llms.openai import (
ChatCompletionToolCallChunk,
@ -267,3 +260,182 @@ def test_regular_tool_finish_reason():
# Verify that finish_reason remains "tool_calls" for regular tools
assert model_response_iterator.converted_response_format_tool is False
assert model_response.choices[0].finish_reason == "tool_calls"
def test_text_only_streaming_has_index_zero():
"""Test that text-only streaming responses have choice index=0"""
chunks = [
{
"type": "message_start",
"message": {
"id": "msg_123",
"type": "message",
"role": "assistant",
"content": [],
"usage": {"input_tokens": 10, "output_tokens": 1},
},
},
{
"type": "content_block_start",
"index": 0,
"content_block": {"type": "text", "text": ""},
},
{
"type": "content_block_delta",
"index": 0,
"delta": {"type": "text_delta", "text": "Hello"},
},
{
"type": "content_block_delta",
"index": 0,
"delta": {"type": "text_delta", "text": " world"},
},
{"type": "content_block_stop", "index": 0},
{
"type": "message_delta",
"delta": {"stop_reason": "end_turn"},
"usage": {"output_tokens": 2},
},
]
iterator = ModelResponseIterator(None, sync_stream=True)
# Check all chunks have choice index=0
for chunk in chunks:
parsed = iterator.chunk_parser(chunk)
if parsed.choices:
assert (
parsed.choices[0].index == 0
), f"Expected index=0, got {parsed.choices[0].index}"
def test_text_and_tool_streaming_has_index_zero():
"""Test that mixed text and tool streaming responses have choice index=0"""
chunks = [
{
"type": "message_start",
"message": {
"id": "msg_123",
"type": "message",
"role": "assistant",
"content": [],
"usage": {"input_tokens": 10, "output_tokens": 1},
},
},
# Reasoning content at index 0
{
"type": "content_block_start",
"index": 0,
"content_block": {"type": "text", "text": ""},
},
{
"type": "content_block_delta",
"index": 0,
"delta": {"type": "text_delta", "text": "I need to search..."},
},
{"type": "content_block_stop", "index": 0},
# Regular content at index 1
{
"type": "content_block_start",
"index": 1,
"content_block": {"type": "text", "text": ""},
},
{
"type": "content_block_delta",
"index": 1,
"delta": {"type": "text_delta", "text": "Let me help you"},
},
{"type": "content_block_stop", "index": 1},
# Tool call at index 2
{
"type": "content_block_start",
"index": 2,
"content_block": {
"type": "tool_use",
"id": "tool_123",
"name": "search",
"input": {},
},
},
{
"type": "content_block_delta",
"index": 2,
"delta": {"type": "input_json_delta", "partial_json": '{"query"'},
},
{
"type": "content_block_delta",
"index": 2,
"delta": {"type": "input_json_delta", "partial_json": ': "test"}'},
},
{"type": "content_block_stop", "index": 2},
{
"type": "message_delta",
"delta": {"stop_reason": "tool_use"},
"usage": {"output_tokens": 10},
},
]
iterator = ModelResponseIterator(None, sync_stream=True)
# Check all chunks have choice index=0 despite different Anthropic indices
for chunk in chunks:
parsed = iterator.chunk_parser(chunk)
if parsed.choices:
assert (
parsed.choices[0].index == 0
), f"Expected index=0 for chunk type {chunk.get('type')}, got {parsed.choices[0].index}"
def test_multiple_tools_streaming_has_index_zero():
"""Test that multiple tool calls all have choice index=0"""
chunks = [
{
"type": "message_start",
"message": {
"id": "msg_123",
"type": "message",
"role": "assistant",
"content": [],
"usage": {"input_tokens": 10, "output_tokens": 1},
},
},
# First tool at index 0
{
"type": "content_block_start",
"index": 0,
"content_block": {
"type": "tool_use",
"id": "tool_1",
"name": "search",
"input": {},
},
},
{"type": "content_block_stop", "index": 0},
# Second tool at index 1
{
"type": "content_block_start",
"index": 1,
"content_block": {
"type": "tool_use",
"id": "tool_2",
"name": "get",
"input": {},
},
},
{"type": "content_block_stop", "index": 1},
{
"type": "message_delta",
"delta": {"stop_reason": "tool_use"},
"usage": {"output_tokens": 5},
},
]
iterator = ModelResponseIterator(None, sync_stream=True)
# All tool chunks should have choice index=0
for chunk in chunks:
parsed = iterator.chunk_parser(chunk)
if parsed.choices:
assert (
parsed.choices[0].index == 0
), f"Expected index=0, got {parsed.choices[0].index}"