From 298a2fd0b975fcb8f365956600773b4bde5c4683 Mon Sep 17 00:00:00 2001 From: ly-wang19 Date: Tue, 23 Jun 2026 18:41:44 +0800 Subject: [PATCH] fix(oci): only emit streaming finish_reason="tool_calls" on the terminal chunk MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses @greptileai's P1. handle_generic_stream_chunk set finish_reason="tool_calls" whenever tool_calls were present, including on intermediate chunks — OCI streams tool calls progressively, so a non-terminal chunk arrives with finishReason=null and toolCalls populated. Per OpenAI streaming semantics finish_reason must be null on non-final deltas; a non-null value there signals premature end-of-stream to callers that stop on finish_reason is not None. Guard the override with `finish_reason is not None` (matching the cohere streaming handler). Adds a regression test for an intermediate chunk (finishReason=null + toolCalls) asserting finish_reason stays None. --- litellm/llms/oci/chat/generic.py | 8 +++++-- .../llms/oci/chat/test_oci_finish_reason.py | 23 +++++++++++++++++++ 2 files changed, 29 insertions(+), 2 deletions(-) diff --git a/litellm/llms/oci/chat/generic.py b/litellm/llms/oci/chat/generic.py index 58555c30c6d..0394f5e001f 100644 --- a/litellm/llms/oci/chat/generic.py +++ b/litellm/llms/oci/chat/generic.py @@ -463,8 +463,12 @@ def handle_generic_stream_chunk(dict_chunk: dict) -> ModelResponseStream: finish_reason: Optional[str] = _normalize_oci_finish_reason( typed_chunk.finishReason ) - # OpenAI semantics: a chunk carrying tool calls reports finish_reason="tool_calls". - if tool_calls: + # OpenAI semantics: the terminal chunk reports finish_reason="tool_calls" when + # tool calls are present. Guard with `finish_reason is not None` so intermediate + # chunks (OCI streams tool calls progressively, with finishReason=null and + # toolCalls populated) keep finish_reason=None — a non-null value on a + # non-terminal chunk would signal premature stream termination to callers. + if finish_reason is not None and tool_calls: finish_reason = "tool_calls" return ModelResponseStream( diff --git a/tests/test_litellm/llms/oci/chat/test_oci_finish_reason.py b/tests/test_litellm/llms/oci/chat/test_oci_finish_reason.py index 41b4b886ded..a8ac5419ab9 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_finish_reason.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_finish_reason.py @@ -145,3 +145,26 @@ def test_generic_stream_chunk_with_tool_calls_reports_tool_calls(): } result = handle_generic_stream_chunk(chunk) assert result.choices[0].finish_reason == "tool_calls" + + +def test_generic_stream_intermediate_chunk_keeps_finish_reason_none(): + # OCI streams tool calls progressively: an intermediate chunk carries + # toolCalls but finishReason=null. finish_reason must stay None on a + # non-terminal chunk — emitting "tool_calls" here would signal premature + # stream termination to callers that stop on finish_reason is not None. + chunk = { + "finishReason": None, + "message": { + "toolCalls": [ + { + "id": "call_0", + "type": "FUNCTION", + "name": "get_weather", + "arguments": "{}", + } + ] + }, + "index": 0, + } + result = handle_generic_stream_chunk(chunk) + assert result.choices[0].finish_reason is None