diff --git a/backend/open_webui/utils/middleware.py b/backend/open_webui/utils/middleware.py
index 56226fc226..9b951c99ad 100644
--- a/backend/open_webui/utils/middleware.py
+++ b/backend/open_webui/utils/middleware.py
@@ -219,6 +219,35 @@ def _split_tool_calls(
return expanded
+def _merge_reasoning_details(details: list, chunk: object) -> None:
+ """Merge reasoning_details delta items into a list in-place.
+
+ Items sharing an integer index are merged: text and summary deltas
+ concatenate when string-valued, other fields overwrite with the latest
+ delta. Items without a usable index are appended in arrival order.
+ """
+ items = chunk if isinstance(chunk, list) else ([chunk] if isinstance(chunk, dict) else [])
+ for item in items:
+ if not isinstance(item, dict):
+ continue
+ idx = item.get('index')
+ existing = None
+ if isinstance(idx, int) and idx >= 0:
+ existing = next((e for e in details if e.get('index') == idx), None)
+ if existing is None:
+ new_entry = dict(item)
+ new_entry.setdefault('type', 'reasoning.text')
+ details.append(new_entry)
+ continue
+ for k, v in item.items():
+ if k in ('text', 'summary'):
+ if isinstance(v, str):
+ base = existing.get(k)
+ existing[k] = (base + v) if isinstance(base, str) else v
+ else:
+ existing[k] = v
+
+
def get_citation_source_from_tool_result(
tool_name: str, tool_params: dict, tool_result: str, tool_id: str = ''
) -> list[dict]:
@@ -523,6 +552,8 @@ def serialize_output(output: list) -> str:
pass
reasoning_content = ''.join(reasoning_parts).strip()
+ if not reasoning_content:
+ continue # no displayable text; item stays in output for round-trip
duration = item.get('duration')
status = item.get('status', 'in_progress')
@@ -3471,7 +3502,35 @@ async def non_streaming_chat_response_handler(response, ctx):
# otherwise generate from response content
response_output = response_data.get('output')
if not response_output:
- response_output = [
+ message_obj = choices[0].get('message', {})
+ reasoning_text = (
+ message_obj.get('reasoning_content')
+ or message_obj.get('reasoning')
+ )
+ reasoning_details = message_obj.get('reasoning_details')
+
+ response_output = []
+
+ if reasoning_text or reasoning_details:
+ r_item = {
+ 'type': 'reasoning',
+ 'id': output_id('r'),
+ 'status': 'completed',
+ 'start_tag': '',
+ 'end_tag': '',
+ 'attributes': {'type': 'reasoning_content'},
+ 'content': [{'type': 'output_text', 'text': reasoning_text}] if reasoning_text else [],
+ 'summary': None,
+ }
+ if reasoning_details:
+ r_item['reasoning_details'] = (
+ reasoning_details
+ if isinstance(reasoning_details, list)
+ else [reasoning_details]
+ )
+ response_output.append(r_item)
+
+ response_output.append(
{
'type': 'message',
'id': output_id('msg'),
@@ -3479,7 +3538,7 @@ async def non_streaming_chat_response_handler(response, ctx):
'role': 'assistant',
'content': [{'type': 'output_text', 'text': content}],
}
- ]
+ )
await event_emitter(
{
@@ -3824,6 +3883,8 @@ async def streaming_chat_response_handler(response, ctx):
# Initialize output: use existing from message if continuing, else create new
existing_output = message.get('output') if message else None
+ _pending_reasoning_details = []
+
if existing_output:
output = existing_output
else:
@@ -4188,9 +4249,14 @@ async def streaming_chat_response_handler(response, ctx):
or delta.get('reasoning')
or delta.get('thinking')
)
+ reasoning_details_chunk = delta.get('reasoning_details')
+
+ # Only create a reasoning item for visible reasoning text.
+ # Details-only deltas (e.g. Gemini encrypted blobs) are
+ # buffered to avoid splitting the assistant message mid-stream.
if reasoning_content:
if not output or output[-1].get('type') != 'reasoning':
- reasoning_item = {
+ output.append({
'type': 'reasoning',
'id': output_id('r'),
'status': 'in_progress',
@@ -4200,23 +4266,37 @@ async def streaming_chat_response_handler(response, ctx):
'content': [],
'summary': None,
'started_at': time.time(),
- }
- output.append(reasoning_item)
- else:
- reasoning_item = output[-1]
+ })
- # Append to reasoning content
+ if reasoning_content:
+ reasoning_item = output[-1]
parts = reasoning_item.get('content', [])
if parts and parts[-1].get('type') == 'output_text':
parts[-1]['text'] += reasoning_content
else:
- reasoning_item['content'] = [
- {
- 'type': 'output_text',
- 'text': reasoning_content,
- }
- ]
+ reasoning_item['content'] = [{'type': 'output_text', 'text': reasoning_content}]
+ # Flush any buffered details-only chunks into this reasoning item.
+ if _pending_reasoning_details:
+ _merge_reasoning_details(
+ reasoning_item.setdefault('reasoning_details', []),
+ _pending_reasoning_details,
+ )
+ _pending_reasoning_details.clear()
+
+ # Accumulate raw structured reasoning_details for provider round-trip.
+ if reasoning_details_chunk:
+ if output and output[-1].get('type') == 'reasoning':
+ _merge_reasoning_details(
+ output[-1].setdefault('reasoning_details', []),
+ reasoning_details_chunk,
+ )
+ else:
+ # Buffer until a safe boundary (end-of-stream or real reasoning text).
+ items = reasoning_details_chunk if isinstance(reasoning_details_chunk, list) else [reasoning_details_chunk]
+ _pending_reasoning_details.extend(items)
+
+ if reasoning_content or reasoning_details_chunk:
data = {'content': serialize_output(full_output())}
if value:
@@ -4430,6 +4510,30 @@ async def streaming_chat_response_handler(response, ctx):
)
reasoning_item['status'] = 'completed'
+ # Flush any buffered reasoning_details that never found a reasoning item.
+ if _pending_reasoning_details:
+ target = next((item for item in output if item.get('type') == 'reasoning'), None)
+ if target is None:
+ target = {
+ 'type': 'reasoning',
+ 'id': output_id('r'),
+ 'status': 'completed',
+ 'start_tag': '',
+ 'end_tag': '',
+ 'attributes': {'type': 'reasoning_content'},
+ 'content': [],
+ 'summary': None,
+ 'started_at': time.time(),
+ 'ended_at': time.time(),
+ 'duration': 0,
+ }
+ output.insert(0, target)
+ _merge_reasoning_details(
+ target.setdefault('reasoning_details', []),
+ _pending_reasoning_details,
+ )
+ _pending_reasoning_details.clear()
+
if response_tool_calls:
tool_calls.append(_split_tool_calls(response_tool_calls))
diff --git a/backend/open_webui/utils/misc.py b/backend/open_webui/utils/misc.py
index b6df292890..0bbb6dada7 100644
--- a/backend/open_webui/utils/misc.py
+++ b/backend/open_webui/utils/misc.py
@@ -159,10 +159,11 @@ def convert_output_to_messages(
pending_tool_calls = []
pending_content = []
pending_reasoning = [] # Only populated when reasoning_format == 'reasoning_content'
+ pending_reasoning_details = None
def flush_pending():
- nonlocal pending_content, pending_tool_calls, pending_reasoning
- if not pending_content and not pending_tool_calls and not pending_reasoning:
+ nonlocal pending_content, pending_tool_calls, pending_reasoning, pending_reasoning_details
+ if not pending_content and not pending_tool_calls and not pending_reasoning and not pending_reasoning_details:
return
message = {
@@ -174,10 +175,14 @@ def convert_output_to_messages(
if pending_reasoning:
message['reasoning_content'] = '\n'.join(pending_reasoning)
+ if pending_reasoning_details:
+ message['reasoning_details'] = pending_reasoning_details
+
messages.append(message)
pending_content = []
pending_tool_calls = []
pending_reasoning = []
+ pending_reasoning_details = None
for item in output:
item_type = item.get('type', '')
@@ -248,7 +253,7 @@ def convert_output_to_messages(
)
elif item_type == 'reasoning':
- if not reasoning_format:
+ if not raw:
continue
reasoning_text = ''
@@ -259,15 +264,28 @@ def convert_output_to_messages(
elif 'text' in part:
reasoning_text += part.get('text', '')
+ raw_details = item.get('reasoning_details')
+
if reasoning_text:
- if reasoning_format == 'think_tags':
- # Ollama: embed in content with the item's original tags
+ if reasoning_format == 'reasoning_content':
+ # llama.cpp: collect for reasoning_content field
+ pending_reasoning.append(reasoning_text)
+ elif not raw_details:
+ # Wrap reasoning in tags for Ollama and any
+ # provider that lacks structured reasoning_details.
start_tag = item.get('start_tag', '')
end_tag = item.get('end_tag', '')
pending_content.append(f'{start_tag}{reasoning_text}{end_tag}')
- elif reasoning_format == 'reasoning_content':
- # llama.cpp: collect for reasoning_content field
- pending_reasoning.append(reasoning_text)
+
+ # Preserve raw structured reasoning_details for provider round-trip
+ if raw_details:
+ if pending_reasoning_details is None:
+ pending_reasoning_details = list(raw_details) if isinstance(raw_details, list) else [raw_details]
+ elif isinstance(pending_reasoning_details, list):
+ if isinstance(raw_details, list):
+ pending_reasoning_details.extend(raw_details)
+ else:
+ pending_reasoning_details.append(raw_details)
elif item_type == 'open_webui:code_interpreter':
# Always include code interpreter content so the LLM knows