diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index ad5f0da8542..f7a81082f80 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -36,6 +36,7 @@ from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM, bedrock_bearer_token from litellm.llms.bedrock.common_utils import ( BedrockError, bedrock_model_is_openai_gpt, + bedrock_runtime_chat_completions_serves_reasoning_inline, split_bedrock_region_path, ) from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler @@ -213,7 +214,12 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]: class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler): - """OpenAI chunk parsing plus the ```` split, tracked per choice index.""" + """OpenAI chunk parsing plus the inline ```` split, tracked per choice index. + + Every chunk echoes the model id litellm sent, so the split engages only when that id's price-map + row carries ``supports_bedrock_runtime_chat_completions_inline_reasoning`` (gpt-oss); a GPT 5.6 or Grok + answer that starts with a literal ```` tag streams as content. + """ def __init__( self, @@ -226,6 +232,8 @@ class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamin def chunk_parser(self, chunk: dict) -> ModelResponseStream: # mutable-ok: BaseModelResponseIterator signature parsed: Final = super().chunk_parser(chunk) + if not bedrock_runtime_chat_completions_serves_reasoning_inline(parsed.model or ""): + return parsed for choice in parsed.choices: next_state, reasoning, content = _split_streamed_content( self._splitters.get(choice.index, ReasoningTagSplitter()), @@ -475,6 +483,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): json_mode=json_mode, ) set_provider_response_headers_in_hidden_params(response, raw_response.headers) + if not bedrock_runtime_chat_completions_serves_reasoning_inline(model): + return response for choice in response.choices: if not isinstance(choice, Choices) or not isinstance(choice.message.content, str): continue diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 7c637bf6724..c21cb61ce60 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -909,6 +909,17 @@ def bedrock_model_is_openai_gpt(model: str) -> bool: return _openai_gpt_version(model) is not None +def bedrock_runtime_chat_completions_serves_reasoning_inline(model: str) -> bool: + """Whether AWS's native Chat Completions writes this model's reasoning inline in the answer text. + + Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_inline_reasoning`` flag (gpt-oss). + A flagged model opens its answer with a ``...`` block instead of a + ``reasoning_content`` field, so litellm splits that block out for it and keeps every other model's + text as sent. + """ + return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_inline_reasoning") + + BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( ( "guardrailConfig", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 8fcf073521e..bbd0d9ab989 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -42320,6 +42320,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -42338,6 +42339,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -42363,6 +42365,7 @@ "output_cost_per_token_flex": 3e-07, "source": "https://aws.amazon.com/bedrock/pricing/", "supports_audio_input": false, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_response_schema": true, "supports_system_messages": true, @@ -42380,6 +42383,7 @@ "output_cost_per_token_flex": 1e-07, "source": "https://aws.amazon.com/bedrock/pricing/", "supports_audio_input": false, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_response_schema": true, "supports_system_messages": true, @@ -48105,6 +48109,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -48123,6 +48128,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -65951,6 +65957,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -65969,6 +65976,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -66219,6 +66227,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -66237,6 +66246,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 8fcf073521e..bbd0d9ab989 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -42320,6 +42320,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -42338,6 +42339,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -42363,6 +42365,7 @@ "output_cost_per_token_flex": 3e-07, "source": "https://aws.amazon.com/bedrock/pricing/", "supports_audio_input": false, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_response_schema": true, "supports_system_messages": true, @@ -42380,6 +42383,7 @@ "output_cost_per_token_flex": 1e-07, "source": "https://aws.amazon.com/bedrock/pricing/", "supports_audio_input": false, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_response_schema": true, "supports_system_messages": true, @@ -48105,6 +48109,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -48123,6 +48128,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -65951,6 +65957,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -65969,6 +65976,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -66219,6 +66227,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -66237,6 +66246,7 @@ "/v1/chat/completions" ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_inline_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index e51ae0453ea..d3843b78ff4 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -1039,6 +1039,9 @@ "supports_audio_output": { "type": "boolean" }, + "supports_bedrock_runtime_chat_completions_inline_reasoning": { + "type": "boolean" + }, "supports_bedrock_runtime_chat_completions_response_format": { "type": "boolean" }, diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 16ae1114402..9d4c0fac094 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -911,7 +911,7 @@ def _stream_chunk(delta, finish_reason=None, index=0): } -def test_streaming_handler_splits_reasoning_deltas_per_choice(): +def test_streaming_handler_splits_reasoning_deltas_per_choice(local_cost_map): handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) first = handler.chunk_parser(_stream_chunk({"role": "assistant", "content": "I think"})) @@ -934,7 +934,7 @@ def _reasoning_of(parsed): return getattr(parsed.choices[0].delta, "reasoning_content", None) -def test_streaming_handler_keeps_split_state_per_choice_index(): +def test_streaming_handler_keeps_split_state_per_choice_index(local_cost_map): handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) opened = handler.chunk_parser(_stream_chunk({"content": "first"}, index=0)) @@ -949,7 +949,7 @@ def test_streaming_handler_keeps_split_state_per_choice_index(): assert not still_reasoning.choices[0].delta.content -def test_streaming_handler_flushes_held_text_on_an_empty_final_delta(): +def test_streaming_handler_flushes_held_text_on_an_empty_final_delta(local_cost_map): handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) held = handler.chunk_parser(_stream_chunk({"content": "taggedHi"}, finish_reason="stop") @@ -1467,3 +1467,76 @@ def test_gpt56_json_object_with_response_schema_goes_to_converse_as_a_json_tool( assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call" assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}} assert "response_format" not in body + + +LITERAL_TAGGED_ANSWER = "not thinking Hello" + + +@pytest.mark.parametrize("model", ["openai.gpt-5.6-sol", "us.xai.grok-4.6"]) +def test_streaming_handler_keeps_a_literal_reasoning_tag_outside_gpt_oss(local_cost_map, model): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + + opened = handler.chunk_parser({**_stream_chunk({"content": "not thinking"}), "model": model}) + assert opened.choices[0].delta.content == "not thinking" + assert _reasoning_of(opened) is None + + closed = handler.chunk_parser({**_stream_chunk({"content": " Hello"}, finish_reason="stop"), "model": model}) + assert closed.choices[0].delta.content == " Hello" + assert _reasoning_of(closed) is None + + +@pytest.mark.parametrize( + "model, expected_content, expected_reasoning", + [ + ("bedrock/global.openai.gpt-5.6-sol", LITERAL_TAGGED_ANSWER, None), + ("bedrock/chat_completions/us.xai.grok-4.6", LITERAL_TAGGED_ANSWER, None), + ("bedrock/chat_completions/openai.gpt-oss-20b-1:0", "Hello", "not thinking"), + ("bedrock/chat_completions/openai.gpt-oss-safeguard-20b", "Hello", "not thinking"), + ], +) +def test_reasoning_tag_split_applies_to_gpt_oss_answers_only( + local_cost_map, fake_aws_env, model, expected_content, expected_reasoning +): + model_id = model.removeprefix("bedrock/").removeprefix("chat_completions/") + requests, client = _recording_client(json=_chat_completion_json(LITERAL_TAGGED_ANSWER, model_id)) + + response = litellm.completion( + model=model, messages=[{"role": "user", "content": "hello"}], client=client, max_tokens=64 + ) + + assert str(requests[0].url).endswith("/openai/v1/chat/completions") + assert response.choices[0].message.content == expected_content + assert getattr(response.choices[0].message, "reasoning_content", None) == expected_reasoning + + +@pytest.mark.parametrize( + "capability_flags, expected_content, expected_reasoning", + [ + ({"supports_bedrock_runtime_chat_completions_inline_reasoning": True}, "Hello", "not thinking"), + ({}, LITERAL_TAGGED_ANSWER, None), + ], +) +def test_reasoning_tag_split_is_read_from_the_cost_map( + monkeypatch, fake_aws_env, capability_flags, expected_content, expected_reasoning +): + model_id = SYNTHETIC_NATIVE_MODEL.removeprefix("chat_completions/") + monkeypatch.setattr(litellm, "model_cost", {model_id: {"litellm_provider": "bedrock_converse", **capability_flags}}) + requests, client = _recording_client(json=_chat_completion_json(LITERAL_TAGGED_ANSWER, model_id)) + + response = litellm.completion( + model=f"bedrock/{SYNTHETIC_NATIVE_MODEL}", + messages=[{"role": "user", "content": "hello"}], + client=client, + max_tokens=64, + ) + + assert str(requests[0].url).endswith("/openai/v1/chat/completions") + assert response.choices[0].message.content == expected_content + assert getattr(response.choices[0].message, "reasoning_content", None) == expected_reasoning + + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + chunk = handler.chunk_parser( + {**_stream_chunk({"content": LITERAL_TAGGED_ANSWER}, finish_reason="stop"), "model": model_id} + ) + assert chunk.choices[0].delta.content == expected_content + assert _reasoning_of(chunk) == expected_reasoning diff --git a/tests/unit/test_utils.py b/tests/unit/test_utils.py index 8a7c9825f0a..58fd2250d2a 100644 --- a/tests/unit/test_utils.py +++ b/tests/unit/test_utils.py @@ -1020,6 +1020,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_web_search": {"type": "boolean"}, "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"}, "supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"}, + "supports_bedrock_runtime_chat_completions_inline_reasoning": {"type": "boolean"}, "supports_url_context": {"type": "boolean"}, "supports_multimodal": {"type": "boolean"}, "uses_embed_content": {"type": "boolean"},