mirror of
https://github.com/open-webui/open-webui.git
synced 2026-10-02 02:12:25 +00:00
fix: max_tokens sent through the API is ignored for Ollama models (#31437)
When an API request to an Ollama model set max_tokens, Open WebUI passed it on in a place Ollama does not read, so Ollama ignored it and replies ran to full length. The limit now reaches Ollama as its own output length setting, so replies stop at the requested length. It also wins over a max_tokens value saved in the model's advanced parameters, as the API docs describe. Chats in the web UI were not affected, since their limit already reached Ollama correctly. Fixes #31432
This commit is contained in:
parent
8081ac299f
commit
583f5a66d2
1 changed files with 5 additions and 4 deletions
|
|
@ -379,10 +379,6 @@ def convert_payload_openai_to_ollama(openai_payload: dict) -> dict:
|
|||
if 'tools' in openai_payload:
|
||||
ollama_payload['tools'] = openai_payload['tools']
|
||||
|
||||
if 'max_tokens' in openai_payload:
|
||||
ollama_payload['num_predict'] = openai_payload['max_tokens']
|
||||
del openai_payload['max_tokens']
|
||||
|
||||
# If there are advanced parameters in the payload, format them in Ollama's options field
|
||||
if openai_payload.get('options'):
|
||||
# Copied before key deletions below so the caller's options stay intact
|
||||
|
|
@ -430,6 +426,11 @@ def convert_payload_openai_to_ollama(openai_payload: dict) -> dict:
|
|||
ollama_options['stop'] = openai_payload.get('stop')
|
||||
ollama_payload['options'] = ollama_options
|
||||
|
||||
if 'max_tokens' in openai_payload:
|
||||
ollama_options = ollama_payload.get('options', {})
|
||||
ollama_options['num_predict'] = openai_payload['max_tokens']
|
||||
ollama_payload['options'] = ollama_options
|
||||
|
||||
if 'metadata' in openai_payload:
|
||||
ollama_payload['metadata'] = openai_payload['metadata']
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue