fix(bedrock): split the <reasoning> tag for gpt-oss only on native Chat Completions (#44432)

* fix(bedrock): split the <reasoning> tag for gpt-oss only on native Chat Completions

The native Chat Completions route moved a leading <reasoning>...</reasoning>
block into reasoning_content for every model, while only gpt-oss writes its
reasoning inline. A GPT 5.6 or Grok answer that starts with a literal
<reasoning> tag lost that text, streaming and non-streaming alike. Both paths
now split only when the model id is gpt-oss.

* fix(bedrock): drive the inline reasoning split from a cost-map flag

---------

Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-08 16:32:07 -07:00 • committed by GitHub
parent 805bb6888b
commit 6a2d3edce3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 123 additions and 5 deletions

View file

@ -36,6 +36,7 @@ from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM, bedrock_bearer_token
from litellm.llms.bedrock.common_utils import (
BedrockError,
bedrock_model_is_openai_gpt,
bedrock_runtime_chat_completions_serves_reasoning_inline,
split_bedrock_region_path,
)
from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler
@ -213,7 +214,12 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]:
class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler):
"""OpenAI chunk parsing plus the ``<reasoning>`` split, tracked per choice index."""
"""OpenAI chunk parsing plus the inline ``<reasoning>`` split, tracked per choice index.
Every chunk echoes the model id litellm sent, so the split engages only when that id's price-map
row carries ``supports_bedrock_runtime_chat_completions_inline_reasoning`` (gpt-oss); a GPT 5.6 or Grok
answer that starts with a literal ``<reasoning>`` tag streams as content.
"""
def __init__(
self,
@ -226,6 +232,8 @@ class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamin
def chunk_parser(self, chunk: dict) -> ModelResponseStream: # mutable-ok: BaseModelResponseIterator signature
parsed: Final = super().chunk_parser(chunk)
if not bedrock_runtime_chat_completions_serves_reasoning_inline(parsed.model or ""):
return parsed
for choice in parsed.choices:
next_state, reasoning, content = _split_streamed_content(
self._splitters.get(choice.index, ReasoningTagSplitter()),
@ -475,6 +483,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
json_mode=json_mode,
)
set_provider_response_headers_in_hidden_params(response, raw_response.headers)
if not bedrock_runtime_chat_completions_serves_reasoning_inline(model):
return response
for choice in response.choices:
if not isinstance(choice, Choices) or not isinstance(choice.message.content, str):
continue

View file

@ -909,6 +909,17 @@ def bedrock_model_is_openai_gpt(model: str) -> bool:
return _openai_gpt_version(model) is not None
def bedrock_runtime_chat_completions_serves_reasoning_inline(model: str) -> bool:
"""Whether AWS's native Chat Completions writes this model's reasoning inline in the answer text.
Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_inline_reasoning`` flag (gpt-oss).
A flagged model opens its answer with a ``<reasoning>...</reasoning>`` block instead of a
``reasoning_content`` field, so litellm splits that block out for it and keeps every other model's
text as sent.
"""
return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_inline_reasoning")
BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
(
"guardrailConfig",

View file

@ -42320,6 +42320,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -42338,6 +42339,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -42363,6 +42365,7 @@
"output_cost_per_token_flex": 3e-07,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_audio_input": false,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_response_schema": true,
"supports_system_messages": true,
@ -42380,6 +42383,7 @@
"output_cost_per_token_flex": 1e-07,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_audio_input": false,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_response_schema": true,
"supports_system_messages": true,
@ -48105,6 +48109,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -48123,6 +48128,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -65951,6 +65957,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -65969,6 +65976,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -66219,6 +66227,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -66237,6 +66246,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,

View file

@ -42320,6 +42320,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -42338,6 +42339,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -42363,6 +42365,7 @@
"output_cost_per_token_flex": 3e-07,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_audio_input": false,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_response_schema": true,
"supports_system_messages": true,
@ -42380,6 +42383,7 @@
"output_cost_per_token_flex": 1e-07,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_audio_input": false,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_response_schema": true,
"supports_system_messages": true,
@ -48105,6 +48109,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -48123,6 +48128,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -65951,6 +65957,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -65969,6 +65976,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -66219,6 +66227,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -66237,6 +66246,7 @@
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_inline_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,

View file

@ -1039,6 +1039,9 @@
"supports_audio_output": {
"type": "boolean"
},
"supports_bedrock_runtime_chat_completions_inline_reasoning": {
"type": "boolean"
},
"supports_bedrock_runtime_chat_completions_response_format": {
"type": "boolean"
},

View file

@ -911,7 +911,7 @@ def _stream_chunk(delta, finish_reason=None, index=0):
}
def test_streaming_handler_splits_reasoning_deltas_per_choice():
def test_streaming_handler_splits_reasoning_deltas_per_choice(local_cost_map):
handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
first = handler.chunk_parser(_stream_chunk({"role": "assistant", "content": "<reasoning>I think"}))
@ -934,7 +934,7 @@ def _reasoning_of(parsed):
return getattr(parsed.choices[0].delta, "reasoning_content", None)
def test_streaming_handler_keeps_split_state_per_choice_index():
def test_streaming_handler_keeps_split_state_per_choice_index(local_cost_map):
handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
opened = handler.chunk_parser(_stream_chunk({"content": "<reasoning>first"}, index=0))
@ -949,7 +949,7 @@ def test_streaming_handler_keeps_split_state_per_choice_index():
assert not still_reasoning.choices[0].delta.content
def test_streaming_handler_flushes_held_text_on_an_empty_final_delta():
def test_streaming_handler_flushes_held_text_on_an_empty_final_delta(local_cost_map):
handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
held = handler.chunk_parser(_stream_chunk({"content": "<reas"}))
@ -1278,7 +1278,7 @@ def test_gpt_oss_streaming_completion_splits_reasoning(local_cost_map, fake_aws_
assert "".join(delta.content or "" for delta in deltas) == "Hi"
def test_streaming_handler_keeps_native_reasoning_next_to_the_tagged_split():
def test_streaming_handler_keeps_native_reasoning_next_to_the_tagged_split(local_cost_map):
handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
parsed = handler.chunk_parser(
_stream_chunk({"reasoning": "native ", "content": "<reasoning>tagged</reasoning>Hi"}, finish_reason="stop")
@ -1467,3 +1467,76 @@ def test_gpt56_json_object_with_response_schema_goes_to_converse_as_a_json_tool(
assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call"
assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}}
assert "response_format" not in body
LITERAL_TAGGED_ANSWER = "<reasoning>not thinking</reasoning> Hello"
@pytest.mark.parametrize("model", ["openai.gpt-5.6-sol", "us.xai.grok-4.6"])
def test_streaming_handler_keeps_a_literal_reasoning_tag_outside_gpt_oss(local_cost_map, model):
handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
opened = handler.chunk_parser({**_stream_chunk({"content": "<reasoning>not thinking"}), "model": model})
assert opened.choices[0].delta.content == "<reasoning>not thinking"
assert _reasoning_of(opened) is None
closed = handler.chunk_parser({**_stream_chunk({"content": "</reasoning> Hello"}, finish_reason="stop"), "model": model})
assert closed.choices[0].delta.content == "</reasoning> Hello"
assert _reasoning_of(closed) is None
@pytest.mark.parametrize(
"model, expected_content, expected_reasoning",
[
("bedrock/global.openai.gpt-5.6-sol", LITERAL_TAGGED_ANSWER, None),
("bedrock/chat_completions/us.xai.grok-4.6", LITERAL_TAGGED_ANSWER, None),
("bedrock/chat_completions/openai.gpt-oss-20b-1:0", "Hello", "not thinking"),
("bedrock/chat_completions/openai.gpt-oss-safeguard-20b", "Hello", "not thinking"),
],
)
def test_reasoning_tag_split_applies_to_gpt_oss_answers_only(
local_cost_map, fake_aws_env, model, expected_content, expected_reasoning
):
model_id = model.removeprefix("bedrock/").removeprefix("chat_completions/")
requests, client = _recording_client(json=_chat_completion_json(LITERAL_TAGGED_ANSWER, model_id))
response = litellm.completion(
model=model, messages=[{"role": "user", "content": "hello"}], client=client, max_tokens=64
)
assert str(requests[0].url).endswith("/openai/v1/chat/completions")
assert response.choices[0].message.content == expected_content
assert getattr(response.choices[0].message, "reasoning_content", None) == expected_reasoning
@pytest.mark.parametrize(
"capability_flags, expected_content, expected_reasoning",
[
({"supports_bedrock_runtime_chat_completions_inline_reasoning": True}, "Hello", "not thinking"),
({}, LITERAL_TAGGED_ANSWER, None),
],
)
def test_reasoning_tag_split_is_read_from_the_cost_map(
monkeypatch, fake_aws_env, capability_flags, expected_content, expected_reasoning
):
model_id = SYNTHETIC_NATIVE_MODEL.removeprefix("chat_completions/")
monkeypatch.setattr(litellm, "model_cost", {model_id: {"litellm_provider": "bedrock_converse", **capability_flags}})
requests, client = _recording_client(json=_chat_completion_json(LITERAL_TAGGED_ANSWER, model_id))
response = litellm.completion(
model=f"bedrock/{SYNTHETIC_NATIVE_MODEL}",
messages=[{"role": "user", "content": "hello"}],
client=client,
max_tokens=64,
)
assert str(requests[0].url).endswith("/openai/v1/chat/completions")
assert response.choices[0].message.content == expected_content
assert getattr(response.choices[0].message, "reasoning_content", None) == expected_reasoning
handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
chunk = handler.chunk_parser(
{**_stream_chunk({"content": LITERAL_TAGGED_ANSWER}, finish_reason="stop"), "model": model_id}
)
assert chunk.choices[0].delta.content == expected_content
assert _reasoning_of(chunk) == expected_reasoning

View file

@ -1020,6 +1020,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_web_search": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_inline_reasoning": {"type": "boolean"},
"supports_url_context": {"type": "boolean"},
"supports_multimodal": {"type": "boolean"},
"uses_embed_content": {"type": "boolean"},