diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index 36f3e875a7e..6d7162f6f36 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -599,6 +599,8 @@ class LiteLLMMessagesToCompletionTransformationHandler: completion_response: Final = await litellm.acompletion(**completion_kwargs) + thinking_disabled = thinking is None or (isinstance(thinking, dict) and thinking.get("type") == "disabled") + if stream: transformed_stream: Final = ANTHROPIC_ADAPTER.translate_completion_output_params_streaming( completion_response, @@ -606,6 +608,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: tool_name_mapping=tool_name_mapping, polyfill_result=polyfill_result, is_async=True, + thinking_disabled=thinking_disabled, ) if transformed_stream is not None: return transformed_stream @@ -615,6 +618,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: cast(ModelResponse, completion_response), tool_name_mapping=tool_name_mapping, polyfill_result=polyfill_result, + thinking_disabled=thinking_disabled, ) if anthropic_response is not None: return anthropic_response @@ -733,6 +737,8 @@ class LiteLLMMessagesToCompletionTransformationHandler: completion_response: Final = litellm.completion(**completion_kwargs) + thinking_disabled = thinking is None or (isinstance(thinking, dict) and thinking.get("type") == "disabled") + if stream: transformed_stream: Final = ANTHROPIC_ADAPTER.translate_completion_output_params_streaming( completion_response, @@ -740,6 +746,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: tool_name_mapping=tool_name_mapping, polyfill_result=polyfill_result, is_async=False, + thinking_disabled=thinking_disabled, ) if transformed_stream is not None: return transformed_stream @@ -749,6 +756,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: cast(ModelResponse, completion_response), tool_name_mapping=tool_name_mapping, polyfill_result=polyfill_result, + thinking_disabled=thinking_disabled, ) if anthropic_response is not None: return anthropic_response diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py index 1660f56378f..8bc4aa57bf7 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py @@ -11,6 +11,7 @@ from typing import ( Final, Literal, Protocol, + cast, get_args, ) @@ -272,7 +273,9 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): sent_first_chunk: bool = False sent_content_block_start: bool = False sent_content_block_finish: bool = False - current_content_block_type: Literal["text", "tool_use", "thinking"] = "text" + current_content_block_type: Literal[ + "text", "tool_use", "thinking", "redacted_thinking" + ] = "text" sent_last_message: bool = False holding_chunk: ContentBlockDelta | None = None holding_stop_reason_chunk: MessageBlockDelta | None = None @@ -287,6 +290,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): applied_edits: list[AppliedEdit] | None = None, compaction_block: CompactionBlock | None = None, iterations_usage: list[UsageIteration] | None = None, + thinking_disabled: bool = False, ): # Wrap the upstream stream so chunks that carry both content and a # finish_reason (fake-streamed providers) are split into two — see @@ -300,6 +304,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): # Synthesized compaction block from compact_20260112 polyfill (streaming). self.compaction_block = compaction_block self.iterations_usage = iterations_usage + self.thinking_disabled = thinking_disabled self.sent_compaction_block: bool = False # Per-phase flags so the compaction block's start/delta/stop events # are emitted (and the public state machine is advanced) in @@ -492,7 +497,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): cache_read_input_tokens=0, ) - def __next__(self): + def __next__(self): # noqa: PLR0915 from .transformation import LiteLLMAnthropicMessagesAdapter try: @@ -551,6 +556,29 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): elif should_start_new_block: self._increment_content_block_index() + is_final_chunk = chunk.choices[0].finish_reason is not None + + # Guard fired in _should_start_new_content_block: a + # non-substantial (empty/role-only) chunk arrived while a + # thinking block is open, and the guard suppressed the block + # TRANSITION (correctly, no spurious text block opens) but the + # chunk would still be translated below into an empty + # text_delta and emitted INSIDE the open thinking block — a + # block-type/delta-type mismatch, the exact class of bug this + # patch series exists to prevent. Suppress the chunk entirely. + # Exclude the finish chunk (it ALSO has should_start_new_block + # == False, per _should_start_new_content_block's own early + # `if chunk.choices[0].finish_reason is not None: return False` + # guard) — it must still flow through to close the block and + # emit message_delta/message_stop, not be silently dropped. + if ( + not should_start_new_block + and not is_final_chunk + and self.current_content_block_type in ("thinking", "redacted_thinking") + and not self._chunk_has_substantial_content(chunk, thinking_disabled=self.thinking_disabled) + ): + continue + # applied_edits only needs to flow to the final message_delta # (when finish_reason is set); skip threading it through every # intermediate chunk. For the hold-and-merge path below, @@ -561,11 +589,15 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): will_merge_into_held = ( self.holding_stop_reason_chunk is not None and getattr(chunk, "usage", None) is not None ) - is_final_chunk = chunk.choices[0].finish_reason is not None processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic( response=chunk, current_content_block_index=self.current_content_block_index, - applied_edits=(self.applied_edits if is_final_chunk and not will_merge_into_held else None), + applied_edits=( + self.applied_edits + if is_final_chunk and not will_merge_into_held + else None + ), + thinking_disabled=self.thinking_disabled, ) # Check if this is a usage chunk and we have a held stop_reason chunk @@ -627,20 +659,24 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): continue if processed_chunk["type"] == "message_delta" and self.sent_content_block_finish is False: - # Queue both the content_block_stop and the message_delta - self.chunk_queue.append( - { - "type": "content_block_stop", - "index": self.current_content_block_index, - } - ) + # Empty responses legitimately have no content block. Only + # close a block if one was actually opened. + if self.sent_content_block_start: + self.chunk_queue.append( + { + "type": "content_block_stop", + "index": self.current_content_block_index, + } + ) self.sent_content_block_finish = True if processed_chunk.get("delta", {}).get("stop_reason") is not None: self.holding_stop_reason_chunk = processed_chunk else: processed_chunk = self._augment_message_delta_usage(processed_chunk) self.chunk_queue.append(processed_chunk) - return self.chunk_queue.popleft() + if self.chunk_queue: + return self.chunk_queue.popleft() + continue elif self.holding_chunk is not None: self.chunk_queue.append(self.holding_chunk) if processed_chunk.get("type") == "message_delta": @@ -672,7 +708,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): # valid Anthropic order (... -> content_block_stop -> # message_delta). Emit ``content_block_stop`` here if # the active content block was not already closed. - if not self.sent_content_block_finish: + if self.sent_content_block_start and not self.sent_content_block_finish: self.chunk_queue.append( { "type": "content_block_stop", @@ -701,7 +737,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): # Anthropic SSE ordering is preserved (content_block_stop -> # message_delta). if self.holding_stop_reason_chunk is not None: - if not self.sent_content_block_finish: + if self.sent_content_block_start and not self.sent_content_block_finish: self.sent_content_block_finish = True self.chunk_queue.append(self._augment_message_delta_usage(self.holding_stop_reason_chunk)) self.holding_stop_reason_chunk = None @@ -779,6 +815,29 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): elif should_start_new_block: self._increment_content_block_index() + is_final_chunk = chunk.choices[0].finish_reason is not None + + # Guard fired in _should_start_new_content_block: a + # non-substantial (empty/role-only) chunk arrived while a + # thinking block is open, and the guard suppressed the block + # TRANSITION (correctly, no spurious text block opens) but the + # chunk would still be translated below into an empty + # text_delta and emitted INSIDE the open thinking block — a + # block-type/delta-type mismatch, the exact class of bug this + # patch series exists to prevent. Suppress the chunk entirely. + # Exclude the finish chunk (it ALSO has should_start_new_block + # == False, per _should_start_new_content_block's own early + # `if chunk.choices[0].finish_reason is not None: return False` + # guard) — it must still flow through to close the block and + # emit message_delta/message_stop, not be silently dropped. + if ( + not should_start_new_block + and not is_final_chunk + and self.current_content_block_type in ("thinking", "redacted_thinking") + and not self._chunk_has_substantial_content(chunk, thinking_disabled=self.thinking_disabled) + ): + continue + # applied_edits only needs to flow to the final message_delta # (when finish_reason is set); skip threading it through every # intermediate chunk. For the hold-and-merge path below, @@ -789,11 +848,15 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): will_merge_into_held = ( self.holding_stop_reason_chunk is not None and getattr(chunk, "usage", None) is not None ) - is_final_chunk = chunk.choices[0].finish_reason is not None processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic( response=chunk, current_content_block_index=self.current_content_block_index, - applied_edits=(self.applied_edits if is_final_chunk and not will_merge_into_held else None), + applied_edits=( + self.applied_edits + if is_final_chunk and not will_merge_into_held + else None + ), + thinking_disabled=self.thinking_disabled, ) # Check if this is a usage chunk and we have a held stop_reason chunk @@ -850,20 +913,24 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): continue if processed_chunk["type"] == "message_delta" and self.sent_content_block_finish is False: - # Queue both the content_block_stop and the holding chunk - self.chunk_queue.append( - { - "type": "content_block_stop", - "index": self.current_content_block_index, - } - ) + # Empty responses legitimately have no content block. Only + # close a block if one was actually opened. + if self.sent_content_block_start: + self.chunk_queue.append( + { + "type": "content_block_stop", + "index": self.current_content_block_index, + } + ) self.sent_content_block_finish = True if processed_chunk.get("delta", {}).get("stop_reason") is not None: self.holding_stop_reason_chunk = processed_chunk else: processed_chunk = self._augment_message_delta_usage(processed_chunk) self.chunk_queue.append(processed_chunk) - return self.chunk_queue.popleft() + if self.chunk_queue: + return self.chunk_queue.popleft() + continue elif self.holding_chunk is not None: # Queue both chunks self.chunk_queue.append(self.holding_chunk) @@ -896,7 +963,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): # valid Anthropic order (... -> content_block_stop -> # message_delta). Emit ``content_block_stop`` here if # the active content block was not already closed. - if not self.sent_content_block_finish: + if self.sent_content_block_start and not self.sent_content_block_finish: self.chunk_queue.append( { "type": "content_block_stop", @@ -930,7 +997,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): # Anthropic SSE ordering is preserved (content_block_stop -> # message_delta). if self.holding_stop_reason_chunk is not None: - if not self.sent_content_block_finish: + if self.sent_content_block_start and not self.sent_content_block_finish: self.sent_content_block_finish = True self.chunk_queue.append(self._augment_message_delta_usage(self.holding_stop_reason_chunk)) self.holding_stop_reason_chunk = None @@ -1016,7 +1083,36 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): delta_type: Final = delta.get("type") if delta_type not in _STREAMING_DELTA_TYPES: return False - return bool(delta.get(_delta_payload_field(delta_type))) + # The membership test above is the runtime guard; the cast tells the type + # checker what it already proved, so the exhaustive match in + # _delta_payload_field keeps its compile-time value. + return bool(delta.get(_delta_payload_field(cast(StreamingContentBlockDeltaType, delta_type)))) + + @staticmethod + def _chunk_has_substantial_content(chunk: "ModelResponseStream", thinking_disabled: bool = False) -> bool: + """Return True when the chunk carries content that should determine or + continue a content block. Delegates to the shared classifier + (ADR-0022) so this check can never diverge from the block-type + classifier or delta emitter's own notion of substantiality — the + root cause of the CTG-85 corrected bug was exactly this kind of + divergence (this function previously used a truthy check while the + classifier used .strip()). + + Remains a @staticmethod with an explicit thinking_disabled parameter + (default False) rather than becoming an instance method, because two + pre-existing tests (test_empty_chunk_is_not_substantial, + test_reasoning_chunk_is_substantial) call it unbound as + AnthropicStreamWrapper._chunk_has_substantial_content(chunk) — converting + to an instance method would break those calls.""" + from .transformation import LiteLLMAnthropicMessagesAdapter + + return ( + LiteLLMAnthropicMessagesAdapter._classify_streaming_chunk( + choices=chunk.choices, # type: ignore + thinking_disabled=thinking_disabled, + ) + != "skip" + ) @staticmethod def _is_blank_delta(chunk: "ModelResponseStream") -> bool: @@ -1055,7 +1151,8 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): block_type, content_block_start, ) = LiteLLMAnthropicMessagesAdapter()._translate_streaming_openai_chunk_to_anthropic_content_block( - choices=chunk.choices + choices=chunk.choices, # type: ignore + thinking_disabled=self.thinking_disabled, ) # Restore original tool name if it was truncated for OpenAI's 64-char limit @@ -1073,6 +1170,12 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): tool_block["name"] = original_name if block_type != self.current_content_block_type: + if ( + block_type == "text" + and self.current_content_block_type in ("thinking", "redacted_thinking") + and not self._chunk_has_substantial_content(chunk, thinking_disabled=self.thinking_disabled) + ): + return False self.current_content_block_type = block_type self.current_content_block_start = content_block_start return True diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index 22f9bfd30ea..a7e1398fb73 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -187,6 +187,7 @@ class AnthropicAdapter: response: ModelResponse, tool_name_mapping: dict[str, str] | None = None, polyfill_result: PolyfillResult | None = None, + thinking_disabled: bool = False, ) -> AnthropicMessagesResponse | None: """ Translate OpenAI response to Anthropic format. @@ -197,11 +198,13 @@ class AnthropicAdapter: Used to restore original names for tools that exceeded OpenAI's 64-char limit. polyfill_result: PolyfillResult from context_management polyfill. + thinking_disabled: When True, suppress reasoning_content → thinking block. """ return LiteLLMAnthropicMessagesAdapter().translate_openai_response_to_anthropic( response=response, tool_name_mapping=tool_name_mapping, polyfill_result=polyfill_result, + thinking_disabled=thinking_disabled, ) def translate_completion_output_params_streaming( @@ -211,6 +214,7 @@ class AnthropicAdapter: tool_name_mapping: dict[str, str] | None = None, polyfill_result: PolyfillResult | None = None, is_async: bool = True, + thinking_disabled: bool = False, ) -> AsyncIterator[bytes] | Iterator[bytes] | None: """ Translate OpenAI streaming response to Anthropic format. @@ -237,6 +241,7 @@ class AnthropicAdapter: applied_edits=applied_edits, compaction_block=compaction_block, iterations_usage=iterations_usage, + thinking_disabled=thinking_disabled, ) # Return the SSE-wrapped version for proper event formatting. if is_async: @@ -535,13 +540,20 @@ class LiteLLMAnthropicMessagesAdapter: self._add_cache_control_if_applicable(content, tool_call, model) tool_calls.append(tool_call) elif content.get("type") == "thinking": - thinking_block = ChatCompletionThinkingBlock( - type="thinking", - thinking=content.get("thinking") or "", - signature=content.get("signature") or "", - cache_control=content.get("cache_control", {}), - ) - thinking_blocks.append(thinking_block) + # Only include thinking blocks that have a real + # signature. Blocks synthesized from flat + # reasoning_content have no signature — passing + # them to Claude causes: + # "signature.str: Input should be a valid string" + # Strip them so multi-turn history stays clean. + if content.get("signature"): + thinking_block = ChatCompletionThinkingBlock( + type="thinking", + thinking=content.get("thinking") or "", + signature=content.get("signature") or "", + cache_control=content.get("cache_control", {}), + ) + thinking_blocks.append(thinking_block) elif content.get("type") == "redacted_thinking": redacted_thinking_block = ChatCompletionRedactedThinkingBlock( type="redacted_thinking", @@ -1132,8 +1144,9 @@ class LiteLLMAnthropicMessagesAdapter: self, choices: list[Choices], tool_name_mapping: dict[str, str] | None = None, + thinking_disabled: bool = False, ) -> list[dict[str, Any]]: - new_content: Final[list[dict[str, Any]]] = [] + new_content: list[dict[str, Any]] = [] for choice in choices: # Handle thinking blocks first if hasattr(choice.message, "thinking_blocks") and choice.message.thinking_blocks: @@ -1156,15 +1169,28 @@ class LiteLLMAnthropicMessagesAdapter: data=str(data_value) if data_value is not None else "", ).model_dump() ) - # Handle reasoning_content when thinking_blocks is not present - elif hasattr(choice.message, "reasoning_content") and choice.message.reasoning_content: - new_content.append( - AnthropicResponseContentBlockThinking( - type="thinking", - thinking=str(choice.message.reasoning_content), - signature=None, - ).model_dump() - ) + # Handle reasoning_content when thinking_blocks is not present. + # Skip if the original request had thinking disabled — a provider + # may still return reasoning_content, but emitting + # a thinking block when the client said thinking=disabled causes + # "Content block is not a thinking block" on the client side. + # Also skip empty or whitespace-only reasoning_content — Anthropic + # rejects thinking blocks with no content ("each thinking block + # must contain thinking") when they are replayed as history. + elif ( + not thinking_disabled + and hasattr(choice.message, "reasoning_content") + and choice.message.reasoning_content + ): + reasoning = str(choice.message.reasoning_content).strip() + if reasoning: + new_content.append( + AnthropicResponseContentBlockThinking( + type="thinking", + thinking=reasoning, + signature=None, + ).model_dump() + ) # Handle text content if choice.message.content is not None: @@ -1296,6 +1322,7 @@ class LiteLLMAnthropicMessagesAdapter: response: ModelResponse, tool_name_mapping: dict[str, str] | None = None, polyfill_result: PolyfillResult | None = None, + thinking_disabled: bool = False, ) -> AnthropicMessagesResponse: """ Translate OpenAI response to Anthropic format. @@ -1306,11 +1333,14 @@ class LiteLLMAnthropicMessagesAdapter: Used to restore original names for tools that exceeded OpenAI's 64-char limit. polyfill_result: PolyfillResult from context_management polyfill. + thinking_disabled: When True, suppress reasoning_content translation + into Anthropic thinking blocks. """ ## translate content block anthropic_content: Final = self._translate_openai_content_to_anthropic( choices=response.choices, tool_name_mapping=tool_name_mapping, + thinking_disabled=thinking_disabled, ) if polyfill_result is not None and polyfill_result.compaction_block is not None: @@ -1349,21 +1379,155 @@ class LiteLLMAnthropicMessagesAdapter: return translated_obj + @staticmethod + def _classify_streaming_chunk( + choices: list["OpenAIStreamingChoice | StreamingChoices"], + thinking_disabled: bool = False, + ) -> Literal["thinking", "redacted_thinking", "tool_use", "text", "skip"]: + """ + Single source of truth for what an OpenAI-format streaming chunk + represents in Anthropic terms. Both the block-type classifier + (_translate_streaming_openai_chunk_to_anthropic_content_block) and the + delta emitter (_translate_streaming_openai_chunk_to_anthropic) MUST + derive their decision from this function's result for the same chunk, + so they can never disagree on the open block's type (ADR-0022, + CTG-85 corrected fix). + + Precedence when multiple signals are present in one chunk: + thinking > tool_use > text > skip. + + "Substantial" reasoning uses .strip() — a whitespace-only + reasoning_content chunk is NOT substantial and returns "skip". + Text content, by contrast, uses a plain truthy check — whitespace IS + meaningful in visible answer text (e.g. a lone " " token between two + words in a streamed response), so a whitespace-only content chunk + still returns "text", not "skip". Applying .strip() to text would + silently drop those tokens. See the inline comments near + `has_substantial_text` below for the full rationale (and the test + `test_classify_whitespace_text_is_still_text_not_skip`). + + Returns "skip" when the chunk carries nothing that should determine or + continue any content block (role-only chunk, whitespace-only reasoning + with no other signal, or a disabled-thinking chunk whose only content is + empty/whitespace reasoning). + """ + for choice in choices: + has_tool_calls = ( + choice.delta.tool_calls is not None + and len(choice.delta.tool_calls) > 0 + and choice.delta.tool_calls[0].function is not None + ) + + # Reasoning signal: thinking_blocks (structured) OR reasoning_content + # (flat string from an OpenAI-compatible provider). Use getattr with a + # default throughout — Delta deletes reasoning_content/thinking_blocks + # entirely when unset, so a direct attribute access can raise + # AttributeError. + reasoning_text = "" + has_structured_thinking_block = False + structured_thinking_block_type: str | None = None + if isinstance(choice, StreamingChoices): + thinking_blocks = getattr(choice.delta, "thinking_blocks", None) or [] + if len(thinking_blocks) > 0: + first_block = thinking_blocks[0] + if first_block.get("type") in ("thinking", "redacted_thinking"): + has_structured_thinking_block = True + structured_thinking_block_type = first_block.get("type") + reasoning_text = str(first_block.get("thinking") or "") + if not has_structured_thinking_block: + reasoning_text = str(getattr(choice.delta, "reasoning_content", "") or "") + + # A structured thinking_block is ALWAYS substantial, regardless of + # whether its thinking/signature text happens to be empty — it + # represents an explicit, structured signal from the provider (e.g. + # a redacted_thinking block, or a signature-only closing chunk for an + # already-open thinking block), which is categorically different + # from a flat, un-structured reasoning_content string that can + # legitimately be pure incidental whitespace. Flat reasoning_content, + # by contrast, is only substantial when it has non-whitespace + # content — this is the actual bug fix (a whitespace-only flat + # reasoning_content chunk must classify as 'skip', not 'thinking' or + # 'text'). + # + # IMPORTANT: do not require a non-empty data/signature field here. + # A redacted_thinking block remains a structured provider signal + # even when its encrypted payload is empty. + has_substantial_reasoning = bool(reasoning_text.strip()) or has_structured_thinking_block + + # IMPORTANT: text content substantiality uses a plain truthy check, + # NOT .strip() — unlike reasoning, whitespace IS meaningful in + # visible answer text (e.g. the space between two words arriving as + # separate streaming tokens, "foo", " ", "bar"). Only reasoning_content + # gets the .strip()-based "is this incidental formatting whitespace" + # treatment; applying the same rule to text would silently drop + # legitimate whitespace tokens from the visible answer. + text_content = str(choice.delta.content or "") + has_substantial_text = bool(text_content) + + if ( + not thinking_disabled + and has_substantial_reasoning + and structured_thinking_block_type == "redacted_thinking" + ): + return "redacted_thinking" + if not thinking_disabled and has_substantial_reasoning: + return "thinking" + if has_tool_calls: + return "tool_use" + if has_substantial_text: + return "text" + # Nothing substantial on this choice — try the next choice (multiple + # choices is rare but the existing functions loop over all of them). + if thinking_disabled and (reasoning_text.strip() or has_structured_thinking_block): + # Thinking disabled but the backend still sent reasoning — this + # chunk carries no client-visible content once suppressed. + continue + return "skip" + def _translate_streaming_openai_chunk_to_anthropic_content_block( - self, choices: list[OpenAIStreamingChoice | StreamingChoices] + self, + choices: list[OpenAIStreamingChoice | StreamingChoices], + thinking_disabled: bool = False, ) -> tuple[ - Literal["text", "tool_use", "thinking"], + Literal["text", "tool_use", "thinking", "redacted_thinking"], "ContentBlockContentBlockDict", ]: from litellm._uuid import uuid from litellm.types.llms.anthropic import TextBlock for choice in choices: - if ( - choice.delta.tool_calls is not None - and len(choice.delta.tool_calls) > 0 - and choice.delta.tool_calls[0].function is not None - ): + block_type = self._classify_streaming_chunk(choices=[choice], thinking_disabled=thinking_disabled) + if block_type == "skip": + continue + + if block_type == "thinking": + if ( + isinstance(choice, StreamingChoices) + and hasattr(choice.delta, "thinking_blocks") + and choice.delta.thinking_blocks + and len(choice.delta.thinking_blocks) > 0 + and choice.delta.thinking_blocks[0].get("type") in ("thinking", "redacted_thinking") + ): + thinking_block = choice.delta.thinking_blocks[0] + thinking = thinking_block.get("thinking") or "" + signature = thinking_block.get("signature") or "" + assert isinstance(thinking, str) + assert isinstance(signature, str) + return "thinking", ChatCompletionThinkingBlock( + type="thinking", thinking=thinking, signature=signature + ) + return "thinking", ChatCompletionThinkingBlock(type="thinking", thinking="", signature="") + + if block_type == "redacted_thinking": + thinking_blocks = getattr(choice.delta, "thinking_blocks", None) or [] + data = str(thinking_blocks[0].get("data") or "") + redacted_block = AnthropicResponseContentBlockRedactedThinking( + type="redacted_thinking", + data=data, + ).model_dump() + return "redacted_thinking", cast("ContentBlockContentBlockDict", redacted_block) + + if block_type == "tool_use": raw_id = choice.delta.tool_calls[0].id or str(uuid.uuid4()) tool_name = choice.delta.tool_calls[0].function.name or "" thought_sig: str | None = None @@ -1377,38 +1541,18 @@ class LiteLLMAnthropicMessagesAdapter: "input": {}, } if thought_sig: - tool_block["provider_specific_fields"] = { - "signature": thought_sig, - } + tool_block["provider_specific_fields"] = {"signature": thought_sig} return "tool_use", cast("ContentBlockContentBlockDict", tool_block) - elif choice.delta.content is not None and len(choice.delta.content) > 0: + + if block_type == "text": return "text", TextBlock(type="text", text="") - elif isinstance(choice, StreamingChoices) and hasattr(choice.delta, "thinking_blocks"): - thinking_blocks = choice.delta.thinking_blocks or [] - if len(thinking_blocks) > 0: - thinking_block = thinking_blocks[0] - if thinking_block["type"] == "thinking": - thinking = thinking_block.get("thinking") or "" - signature = thinking_block.get("signature") or "" - - assert isinstance(thinking, str) - assert isinstance(signature, str) - - return "thinking", ChatCompletionThinkingBlock( - type="thinking", thinking=thinking, signature=signature - ) - # OpenAI-compatible reasoning backends (e.g. vLLM/SGLang reasoning - # parsers) populate ``reasoning_content`` without ``thinking_blocks``. - # ``Delta`` deletes the ``thinking_blocks`` attribute when unset, so the - # branch above is skipped entirely; open a ``thinking`` block here so the - # matching ``thinking_delta`` stream is not emitted into a text block. - elif isinstance(choice, StreamingChoices) and getattr(choice.delta, "reasoning_content", None): - return "thinking", ChatCompletionThinkingBlock(type="thinking", thinking="", signature="") return "text", TextBlock(type="text", text="") def _translate_streaming_openai_chunk_to_anthropic( - self, choices: list[OpenAIStreamingChoice | StreamingChoices] + self, + choices: list[OpenAIStreamingChoice | StreamingChoices], + thinking_disabled: bool = False, ) -> tuple[ StreamingContentBlockDeltaType, ContentTextBlockDelta | ContentJsonBlockDelta | ContentThinkingBlockDelta | ContentThinkingSignatureBlockDelta, @@ -1417,32 +1561,41 @@ class LiteLLMAnthropicMessagesAdapter: reasoning_content: str = "" reasoning_signature: str = "" partial_json: str | None = None + for choice in choices: - if choice.delta.content is not None and len(choice.delta.content) > 0: - text += choice.delta.content - if choice.delta.tool_calls: - partial_json = "" - for tool in choice.delta.tool_calls: - if tool.function is not None and tool.function.arguments is not None: - partial_json = (partial_json or "") + tool.function.arguments - elif isinstance(choice, StreamingChoices) and hasattr(choice.delta, "thinking_blocks"): - thinking_blocks = choice.delta.thinking_blocks or [] - if len(thinking_blocks) > 0: - for thinking_block in thinking_blocks: - if thinking_block["type"] == "thinking": - thinking = thinking_block.get("thinking") or "" - signature = thinking_block.get("signature") or "" + block_type = self._classify_streaming_chunk(choices=[choice], thinking_disabled=thinking_disabled) + if block_type == "skip": + continue - assert isinstance(thinking, str) - assert isinstance(signature, str) + if block_type == "thinking": + if ( + isinstance(choice, StreamingChoices) + and hasattr(choice.delta, "thinking_blocks") + and choice.delta.thinking_blocks + and len(choice.delta.thinking_blocks) > 0 + ): + for thinking_block in choice.delta.thinking_blocks: + if thinking_block.get("type") in ("thinking", "redacted_thinking"): + reasoning_content += str(thinking_block.get("thinking") or "") + reasoning_signature += str(thinking_block.get("signature") or "") + elif isinstance(choice, StreamingChoices) and getattr(choice.delta, "reasoning_content", None): + reasoning_content += str(choice.delta.reasoning_content) - reasoning_content += thinking - reasoning_signature += signature - # Handle reasoning_content when thinking_blocks is not present - # This handles providers like OpenRouter that return reasoning_content - elif isinstance(choice, StreamingChoices) and hasattr(choice.delta, "reasoning_content"): - if choice.delta.reasoning_content is not None: - reasoning_content += choice.delta.reasoning_content + elif block_type == "redacted_thinking": + # Redacted thinking is carried wholly in content_block_start; + # Anthropic defines no redacted-thinking delta type. + continue + + elif block_type == "tool_use": + if choice.delta.tool_calls: + partial_json = partial_json or "" + for tool in choice.delta.tool_calls: + if tool.function is not None and tool.function.arguments is not None: + partial_json += tool.function.arguments + + elif block_type == "text": + if choice.delta.content is not None and len(choice.delta.content) > 0: + text += choice.delta.content if partial_json is not None: return "input_json_delta", ContentJsonBlockDelta(type="input_json_delta", partial_json=partial_json) @@ -1460,6 +1613,7 @@ class LiteLLMAnthropicMessagesAdapter: response: ModelResponse, current_content_block_index: int, applied_edits: list[AppliedEdit] | None = None, + thinking_disabled: bool = False, ) -> ContentBlockDelta | MessageBlockDelta: ## base case - final chunk w/ finish reason if response.choices[0].finish_reason is not None: @@ -1487,7 +1641,10 @@ class LiteLLMAnthropicMessagesAdapter: ( type_of_content, content_block_delta, - ) = self._translate_streaming_openai_chunk_to_anthropic(choices=response.choices) + ) = self._translate_streaming_openai_chunk_to_anthropic( + choices=response.choices, # type: ignore + thinking_disabled=thinking_disabled, + ) return ContentBlockDelta( type="content_block_delta", index=current_content_block_index,