From 0a7c50c2380d307c730fe4e9f7ccd1140fba83a8 Mon Sep 17 00:00:00 2001 From: sharziki Date: Thu, 14 May 2026 21:06:32 -0400 Subject: [PATCH] fix(ollama): extract thinking field from /api/generate response Ollama's /api/generate endpoint returns a top-level 'thinking' field for reasoning models (Qwen3, DeepSeek-R1) but the non-streaming transform_response() only parsed XML tags from the response text. This caused reasoning_content to always be null. Check for the 'thinking' field first, matching the pattern already used in the /api/chat path and the streaming iterator. Fixes #27956 --- litellm/llms/ollama/completion/transformation.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/litellm/llms/ollama/completion/transformation.py b/litellm/llms/ollama/completion/transformation.py index 32981776753..cb2d749dab1 100644 --- a/litellm/llms/ollama/completion/transformation.py +++ b/litellm/llms/ollama/completion/transformation.py @@ -370,7 +370,12 @@ class OllamaConfig(BaseConfig): response_text = response_json.get("response", "") content = None reasoning_content = None - if response_text is not None and isinstance(response_text, str): + # Check for top-level "thinking" field (Ollama returns this for + # reasoning models like Qwen3/DeepSeek-R1 via /api/generate) + if "thinking" in response_json: + reasoning_content = response_json["thinking"] + content = response_text + elif response_text is not None and isinstance(response_text, str): reasoning_content, content = _parse_content_for_reasoning(response_text) else: content = response_text # type: ignore