From 5313a9c67f59a16809462b77d8379e0d4075416c Mon Sep 17 00:00:00 2001 From: nehaaprasaad Date: Thu, 7 May 2026 07:56:42 +0530 Subject: [PATCH] fix(cache): align Anthropic reasoning_content with live response on replay --- .../convert_dict_to_response.py | 5 +- .../test_convert_dict_to_chat_completion.py | 46 ++++++++++++++++++- 2 files changed, 47 insertions(+), 4 deletions(-) diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 5fd42fe0d36..b0ba237c627 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -595,10 +595,9 @@ def convert_to_model_response_object( # noqa: PLR0915 thinking_blocks = choice["message"]["thinking_blocks"] provider_specific_fields["thinking_blocks"] = thinking_blocks + # Align with live provider responses: reasoning_content is message-level only. if reasoning_content: - provider_specific_fields["reasoning_content"] = ( - reasoning_content - ) + provider_specific_fields.pop("reasoning_content", None) message = Message( content=content, diff --git a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py index 66a1a4d74af..944ee275369 100644 --- a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py +++ b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py @@ -902,7 +902,51 @@ def test_convert_to_model_response_object_with_thinking_content(): resp: ModelResponse = convert_to_model_response_object(**args) assert resp is not None - assert resp.choices[0].message.reasoning_content is not None + msg = resp.choices[0].message + assert msg.reasoning_content is not None + assert "reasoning_content" not in (msg.provider_specific_fields or {}) + + +def test_convert_to_model_response_object_strips_reasoning_content_from_provider_specific_fields(): + """ + Stale or merged provider_specific_fields must not duplicate top-level reasoning_content. + Duplication breaks disk cache keys on later turns (see Anthropic extended thinking). + """ + duplicate = "same reasoning" + args = { + "response_object": { + "id": "chatcmpl-dup-reasoning", + "created": 1741057687, + "model": "claude-4-sonnet-20250514", + "object": "chat.completion", + "choices": [ + { + "finish_reason": "stop", + "index": 0, + "message": { + "content": "Answer.", + "role": "assistant", + "reasoning_content": duplicate, + "provider_specific_fields": { + "citations": None, + "reasoning_content": duplicate, + "thinking_blocks": [], + }, + }, + } + ], + "usage": { + "completion_tokens": 1, + "prompt_tokens": 1, + "total_tokens": 2, + }, + }, + "model_response_object": ModelResponse(), + } + resp = convert_to_model_response_object(**args) + msg = resp.choices[0].message + assert msg.reasoning_content == duplicate + assert "reasoning_content" not in (msg.provider_specific_fields or {}) def test_convert_to_model_response_object_with_empty_error_object():