diff --git a/backend/open_webui/utils/middleware.py b/backend/open_webui/utils/middleware.py index 56226fc226..9b951c99ad 100644 --- a/backend/open_webui/utils/middleware.py +++ b/backend/open_webui/utils/middleware.py @@ -219,6 +219,35 @@ def _split_tool_calls( return expanded +def _merge_reasoning_details(details: list, chunk: object) -> None: + """Merge reasoning_details delta items into a list in-place. + + Items sharing an integer index are merged: text and summary deltas + concatenate when string-valued, other fields overwrite with the latest + delta. Items without a usable index are appended in arrival order. + """ + items = chunk if isinstance(chunk, list) else ([chunk] if isinstance(chunk, dict) else []) + for item in items: + if not isinstance(item, dict): + continue + idx = item.get('index') + existing = None + if isinstance(idx, int) and idx >= 0: + existing = next((e for e in details if e.get('index') == idx), None) + if existing is None: + new_entry = dict(item) + new_entry.setdefault('type', 'reasoning.text') + details.append(new_entry) + continue + for k, v in item.items(): + if k in ('text', 'summary'): + if isinstance(v, str): + base = existing.get(k) + existing[k] = (base + v) if isinstance(base, str) else v + else: + existing[k] = v + + def get_citation_source_from_tool_result( tool_name: str, tool_params: dict, tool_result: str, tool_id: str = '' ) -> list[dict]: @@ -523,6 +552,8 @@ def serialize_output(output: list) -> str: pass reasoning_content = ''.join(reasoning_parts).strip() + if not reasoning_content: + continue # no displayable text; item stays in output for round-trip duration = item.get('duration') status = item.get('status', 'in_progress') @@ -3471,7 +3502,35 @@ async def non_streaming_chat_response_handler(response, ctx): # otherwise generate from response content response_output = response_data.get('output') if not response_output: - response_output = [ + message_obj = choices[0].get('message', {}) + reasoning_text = ( + message_obj.get('reasoning_content') + or message_obj.get('reasoning') + ) + reasoning_details = message_obj.get('reasoning_details') + + response_output = [] + + if reasoning_text or reasoning_details: + r_item = { + 'type': 'reasoning', + 'id': output_id('r'), + 'status': 'completed', + 'start_tag': '', + 'end_tag': '', + 'attributes': {'type': 'reasoning_content'}, + 'content': [{'type': 'output_text', 'text': reasoning_text}] if reasoning_text else [], + 'summary': None, + } + if reasoning_details: + r_item['reasoning_details'] = ( + reasoning_details + if isinstance(reasoning_details, list) + else [reasoning_details] + ) + response_output.append(r_item) + + response_output.append( { 'type': 'message', 'id': output_id('msg'), @@ -3479,7 +3538,7 @@ async def non_streaming_chat_response_handler(response, ctx): 'role': 'assistant', 'content': [{'type': 'output_text', 'text': content}], } - ] + ) await event_emitter( { @@ -3824,6 +3883,8 @@ async def streaming_chat_response_handler(response, ctx): # Initialize output: use existing from message if continuing, else create new existing_output = message.get('output') if message else None + _pending_reasoning_details = [] + if existing_output: output = existing_output else: @@ -4188,9 +4249,14 @@ async def streaming_chat_response_handler(response, ctx): or delta.get('reasoning') or delta.get('thinking') ) + reasoning_details_chunk = delta.get('reasoning_details') + + # Only create a reasoning item for visible reasoning text. + # Details-only deltas (e.g. Gemini encrypted blobs) are + # buffered to avoid splitting the assistant message mid-stream. if reasoning_content: if not output or output[-1].get('type') != 'reasoning': - reasoning_item = { + output.append({ 'type': 'reasoning', 'id': output_id('r'), 'status': 'in_progress', @@ -4200,23 +4266,37 @@ async def streaming_chat_response_handler(response, ctx): 'content': [], 'summary': None, 'started_at': time.time(), - } - output.append(reasoning_item) - else: - reasoning_item = output[-1] + }) - # Append to reasoning content + if reasoning_content: + reasoning_item = output[-1] parts = reasoning_item.get('content', []) if parts and parts[-1].get('type') == 'output_text': parts[-1]['text'] += reasoning_content else: - reasoning_item['content'] = [ - { - 'type': 'output_text', - 'text': reasoning_content, - } - ] + reasoning_item['content'] = [{'type': 'output_text', 'text': reasoning_content}] + # Flush any buffered details-only chunks into this reasoning item. + if _pending_reasoning_details: + _merge_reasoning_details( + reasoning_item.setdefault('reasoning_details', []), + _pending_reasoning_details, + ) + _pending_reasoning_details.clear() + + # Accumulate raw structured reasoning_details for provider round-trip. + if reasoning_details_chunk: + if output and output[-1].get('type') == 'reasoning': + _merge_reasoning_details( + output[-1].setdefault('reasoning_details', []), + reasoning_details_chunk, + ) + else: + # Buffer until a safe boundary (end-of-stream or real reasoning text). + items = reasoning_details_chunk if isinstance(reasoning_details_chunk, list) else [reasoning_details_chunk] + _pending_reasoning_details.extend(items) + + if reasoning_content or reasoning_details_chunk: data = {'content': serialize_output(full_output())} if value: @@ -4430,6 +4510,30 @@ async def streaming_chat_response_handler(response, ctx): ) reasoning_item['status'] = 'completed' + # Flush any buffered reasoning_details that never found a reasoning item. + if _pending_reasoning_details: + target = next((item for item in output if item.get('type') == 'reasoning'), None) + if target is None: + target = { + 'type': 'reasoning', + 'id': output_id('r'), + 'status': 'completed', + 'start_tag': '', + 'end_tag': '', + 'attributes': {'type': 'reasoning_content'}, + 'content': [], + 'summary': None, + 'started_at': time.time(), + 'ended_at': time.time(), + 'duration': 0, + } + output.insert(0, target) + _merge_reasoning_details( + target.setdefault('reasoning_details', []), + _pending_reasoning_details, + ) + _pending_reasoning_details.clear() + if response_tool_calls: tool_calls.append(_split_tool_calls(response_tool_calls)) diff --git a/backend/open_webui/utils/misc.py b/backend/open_webui/utils/misc.py index b6df292890..0bbb6dada7 100644 --- a/backend/open_webui/utils/misc.py +++ b/backend/open_webui/utils/misc.py @@ -159,10 +159,11 @@ def convert_output_to_messages( pending_tool_calls = [] pending_content = [] pending_reasoning = [] # Only populated when reasoning_format == 'reasoning_content' + pending_reasoning_details = None def flush_pending(): - nonlocal pending_content, pending_tool_calls, pending_reasoning - if not pending_content and not pending_tool_calls and not pending_reasoning: + nonlocal pending_content, pending_tool_calls, pending_reasoning, pending_reasoning_details + if not pending_content and not pending_tool_calls and not pending_reasoning and not pending_reasoning_details: return message = { @@ -174,10 +175,14 @@ def convert_output_to_messages( if pending_reasoning: message['reasoning_content'] = '\n'.join(pending_reasoning) + if pending_reasoning_details: + message['reasoning_details'] = pending_reasoning_details + messages.append(message) pending_content = [] pending_tool_calls = [] pending_reasoning = [] + pending_reasoning_details = None for item in output: item_type = item.get('type', '') @@ -248,7 +253,7 @@ def convert_output_to_messages( ) elif item_type == 'reasoning': - if not reasoning_format: + if not raw: continue reasoning_text = '' @@ -259,15 +264,28 @@ def convert_output_to_messages( elif 'text' in part: reasoning_text += part.get('text', '') + raw_details = item.get('reasoning_details') + if reasoning_text: - if reasoning_format == 'think_tags': - # Ollama: embed in content with the item's original tags + if reasoning_format == 'reasoning_content': + # llama.cpp: collect for reasoning_content field + pending_reasoning.append(reasoning_text) + elif not raw_details: + # Wrap reasoning in tags for Ollama and any + # provider that lacks structured reasoning_details. start_tag = item.get('start_tag', '') end_tag = item.get('end_tag', '') pending_content.append(f'{start_tag}{reasoning_text}{end_tag}') - elif reasoning_format == 'reasoning_content': - # llama.cpp: collect for reasoning_content field - pending_reasoning.append(reasoning_text) + + # Preserve raw structured reasoning_details for provider round-trip + if raw_details: + if pending_reasoning_details is None: + pending_reasoning_details = list(raw_details) if isinstance(raw_details, list) else [raw_details] + elif isinstance(pending_reasoning_details, list): + if isinstance(raw_details, list): + pending_reasoning_details.extend(raw_details) + else: + pending_reasoning_details.append(raw_details) elif item_type == 'open_webui:code_interpreter': # Always include code interpreter content so the LLM knows