feat: pass the provider's finish reason to outlet filters (#32079)

Outlet filters see the finished reply with its content and token usage, but had no way to tell whether the model ended on its own, ran into the token limit or stopped to call a tool, short of reading every chunk in a stream filter. For OpenAI-compatible and Ollama models, the assistant message handed to outlet now carries finish_reason with the value the provider reported on the last model call of the turn, for streaming and non-streaming replies. Ollama replies cut off by the token limit now report length as well, where they always said stop before.
This commit is contained in:
Classic298 2026-10-08 17:36:47 +02:00 • committed by GitHub
parent ce18eca340
commit 039867feba
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 19 additions and 0 deletions

View file

@ -3689,6 +3689,8 @@ def update_assistant_message_from_stream(assistant_message, raw):
assistant_message['usage'] = merge_usage(assistant_message.get('usage'), raw_usage) assistant_message['usage'] = merge_usage(assistant_message.get('usage'), raw_usage)
for choice in data.get('choices', []): for choice in data.get('choices', []):
if choice.get('finish_reason'):
assistant_message['finish_reason'] = choice['finish_reason']
delta = choice.get('delta', {}) or {} delta = choice.get('delta', {}) or {}
content = delta.get('content') content = delta.get('content')
reasoning_content = delta.get('reasoning_content') or delta.get('reasoning') or delta.get('thinking') reasoning_content = delta.get('reasoning_content') or delta.get('reasoning') or delta.get('thinking')
@ -4066,6 +4068,7 @@ async def outlet_filter_handler(ctx):
if not message_list: if not message_list:
return return
finish_reason = (ctx.get('assistant_message') or {}).get('finish_reason')
outlet_data = { outlet_data = {
'model': model_id, 'model': model_id,
'messages': [ 'messages': [
@ -4079,6 +4082,7 @@ async def outlet_filter_handler(ctx):
**({'output': copy.deepcopy(m['output'])} if m.get('output') else {}), **({'output': copy.deepcopy(m['output'])} if m.get('output') else {}),
**({'usage': m['usage']} if m.get('usage') else {}), **({'usage': m['usage']} if m.get('usage') else {}),
**({'sources': m['sources']} if m.get('sources') else {}), **({'sources': m['sources']} if m.get('sources') else {}),
**({'finish_reason': finish_reason} if finish_reason and m.get('id') == message_id else {}),
} }
for m in message_list for m in message_list
], ],
@ -4307,6 +4311,7 @@ async def non_streaming_chat_response_handler(response, ctx):
# Save message in the database # Save message in the database
usage = normalize_usage(response_data.get('usage', {}) or {}) usage = normalize_usage(response_data.get('usage', {}) or {})
finish_reason = choices[0].get('finish_reason') if choices else None
if save_to_chat: if save_to_chat:
await Chats.upsert_message_to_chat_by_id_and_message_id( await Chats.upsert_message_to_chat_by_id_and_message_id(
@ -4326,6 +4331,7 @@ async def non_streaming_chat_response_handler(response, ctx):
'content': content, 'content': content,
'output': response_output, 'output': response_output,
**({'usage': usage} if usage else {}), **({'usage': usage} if usage else {}),
**({'finish_reason': finish_reason} if finish_reason else {}),
} }
await outlet_filter_handler(ctx) await outlet_filter_handler(ctx)
await background_tasks_handler(ctx) await background_tasks_handler(ctx)
@ -4361,10 +4367,12 @@ async def non_streaming_chat_response_handler(response, ctx):
content = choices[0].get('message', {}).get('content') if choices else '' content = choices[0].get('message', {}).get('content') if choices else ''
if ENABLE_API_OUTLET_FILTERS and (content or output): if ENABLE_API_OUTLET_FILTERS and (content or output):
usage = normalize_usage(response_data.get('usage', {}) or {}) usage = normalize_usage(response_data.get('usage', {}) or {})
finish_reason = choices[0].get('finish_reason') if choices else None
ctx['assistant_message'] = { ctx['assistant_message'] = {
**({'content': content} if content else {}), **({'content': content} if content else {}),
**({'output': output} if output else {}), **({'output': output} if output else {}),
**({'usage': usage} if usage else {}), **({'usage': usage} if usage else {}),
**({'finish_reason': finish_reason} if finish_reason else {}),
} }
await outlet_filter_handler(ctx) await outlet_filter_handler(ctx)
@ -4768,6 +4776,7 @@ async def streaming_chat_response_handler(response, ctx):
content_parts = [] content_parts = []
usage = None usage = None
finish_reason = None
last_response_id = None last_response_id = None
def full_output(): def full_output():
@ -4867,6 +4876,7 @@ async def streaming_chat_response_handler(response, ctx):
async def stream_body_handler(response, form_data): async def stream_body_handler(response, form_data):
nonlocal usage nonlocal usage
nonlocal finish_reason
nonlocal output nonlocal output
nonlocal prior_output nonlocal prior_output
nonlocal last_response_id nonlocal last_response_id
@ -5211,6 +5221,9 @@ async def streaming_chat_response_handler(response, ctx):
) )
continue continue
if choices[0].get('finish_reason'):
finish_reason = choices[0]['finish_reason']
delta = choices[0].get('delta', {}) delta = choices[0].get('delta', {})
delta_type = 'content' delta_type = 'content'
@ -6513,6 +6526,7 @@ async def streaming_chat_response_handler(response, ctx):
else ''.join(content_parts) or get_output_text(current_output), else ''.join(content_parts) or get_output_text(current_output),
'output': current_output, 'output': current_output,
**({'usage': usage} if usage else {}), **({'usage': usage} if usage else {}),
**({'finish_reason': finish_reason} if finish_reason else {}),
} }
await outlet_filter_handler(ctx) await outlet_filter_handler(ctx)
await background_tasks_handler(ctx) await background_tasks_handler(ctx)

View file

@ -224,6 +224,8 @@ def convert_response_ollama_to_openai(ollama_response: dict) -> dict:
response = openai_chat_completion_message_template( response = openai_chat_completion_message_template(
model, message_content, reasoning_content, openai_tool_calls, usage model, message_content, reasoning_content, openai_tool_calls, usage
) )
if not openai_tool_calls and ollama_response.get('done_reason'):
response['choices'][0]['finish_reason'] = ollama_response['done_reason']
return response return response
@ -247,6 +249,7 @@ async def convert_streaming_response_ollama_to_openai(ollama_streaming_response)
has_tool_calls = True has_tool_calls = True
done = data.get('done', False) done = data.get('done', False)
done_reason = data.get('done_reason')
usage = None usage = None
if done: if done:
@ -263,6 +266,8 @@ async def convert_streaming_response_ollama_to_openai(ollama_streaming_response)
if done and has_tool_calls: if done and has_tool_calls:
data['choices'][0]['finish_reason'] = 'tool_calls' data['choices'][0]['finish_reason'] = 'tool_calls'
elif done_reason:
data['choices'][0]['finish_reason'] = done_reason
line = f'data: {JSONCodec.dumps(data)}\n\n' line = f'data: {JSONCodec.dumps(data)}\n\n'
yield line yield line