From 2fc99ee00d8f21b56e13d4a8cf4e811509e812fc Mon Sep 17 00:00:00 2001 From: Stephen Chin <1290231+steveonjava@users.noreply.github.com> Date: Thu, 20 Aug 2026 05:49:21 -0700 Subject: [PATCH 1/3] fix(chatgpt): forward prompt cache key --- litellm/llms/chatgpt/responses/transformation.py | 1 + .../chatgpt/responses/test_chatgpt_responses_transformation.py | 2 ++ 2 files changed, 3 insertions(+) diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index b96e06be3d8..dfb554929fa 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -100,6 +100,7 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): "tools", "tool_choice", "reasoning", + "prompt_cache_key", "previous_response_id", "truncation", } diff --git a/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py b/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py index 8e0415d50de..7549da238dc 100644 --- a/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py +++ b/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py @@ -128,6 +128,7 @@ class TestChatGPTResponsesAPITransformation: "max_output_tokens": 123, "stream_options": {"include_usage": True}, # supported and should be preserved + "prompt_cache_key": "session_123", "truncation": "auto", "previous_response_id": "resp_123", "reasoning": {"effort": "medium"}, @@ -146,6 +147,7 @@ class TestChatGPTResponsesAPITransformation: assert "max_output_tokens" not in request assert "stream_options" not in request + assert request["prompt_cache_key"] == "session_123" assert request["truncation"] == "auto" assert request["previous_response_id"] == "resp_123" assert request["reasoning"] == {"effort": "medium"} From 26a2ffea0e520b0b28819b376dc88fbf6f4d7ebc Mon Sep 17 00:00:00 2001 From: Stephen Chin <1290231+steveonjava@users.noreply.github.com> Date: Thu, 20 Aug 2026 06:20:50 -0700 Subject: [PATCH 2/3] fix(chatgpt): forward native compaction config --- litellm/llms/chatgpt/responses/transformation.py | 1 + .../responses/test_chatgpt_responses_transformation.py | 8 ++++---- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index dfb554929fa..e9bc95e88b1 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -101,6 +101,7 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): "tool_choice", "reasoning", "prompt_cache_key", + "context_management", "previous_response_id", "truncation", } diff --git a/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py b/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py index 7549da238dc..0d083cbab6f 100644 --- a/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py +++ b/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py @@ -121,14 +121,12 @@ class TestChatGPTResponsesAPITransformation: "user": "user_123", "temperature": 0.2, "top_p": 0.9, - "context_management": [ - {"type": "compaction", "compact_threshold": 200000} - ], "metadata": {"foo": "bar"}, "max_output_tokens": 123, "stream_options": {"include_usage": True}, # supported and should be preserved "prompt_cache_key": "session_123", + "context_management": [{"type": "compaction", "compact_threshold": 200000}], "truncation": "auto", "previous_response_id": "resp_123", "reasoning": {"effort": "medium"}, @@ -142,12 +140,14 @@ class TestChatGPTResponsesAPITransformation: assert "user" not in request assert "temperature" not in request assert "top_p" not in request - assert "context_management" not in request assert "metadata" not in request assert "max_output_tokens" not in request assert "stream_options" not in request assert request["prompt_cache_key"] == "session_123" + assert request["context_management"] == [ + {"type": "compaction", "compact_threshold": 200000} + ] assert request["truncation"] == "auto" assert request["previous_response_id"] == "resp_123" assert request["reasoning"] == {"effort": "medium"} From afaa5c82e694c96a96ee57a46e2586b3f8adc1b3 Mon Sep 17 00:00:00 2001 From: Stephen Chin Date: Thu, 20 Aug 2026 21:24:06 +0000 Subject: [PATCH 3/3] fix(chatgpt): keep compaction config filtered I removed context_management from the ChatGPT Responses allowlist after\nauthenticated GPT-5.6 probes did not establish normal endpoint support.\nprompt_cache_key remains forwarded, and the regression keeps unsupported\nChatGPT request fields filtered. --- litellm/llms/chatgpt/responses/transformation.py | 1 - .../responses/test_chatgpt_responses_transformation.py | 8 ++++---- 2 files changed, 4 insertions(+), 5 deletions(-) diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index e9bc95e88b1..dfb554929fa 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -101,7 +101,6 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): "tool_choice", "reasoning", "prompt_cache_key", - "context_management", "previous_response_id", "truncation", } diff --git a/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py b/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py index 0d083cbab6f..7549da238dc 100644 --- a/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py +++ b/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py @@ -121,12 +121,14 @@ class TestChatGPTResponsesAPITransformation: "user": "user_123", "temperature": 0.2, "top_p": 0.9, + "context_management": [ + {"type": "compaction", "compact_threshold": 200000} + ], "metadata": {"foo": "bar"}, "max_output_tokens": 123, "stream_options": {"include_usage": True}, # supported and should be preserved "prompt_cache_key": "session_123", - "context_management": [{"type": "compaction", "compact_threshold": 200000}], "truncation": "auto", "previous_response_id": "resp_123", "reasoning": {"effort": "medium"}, @@ -140,14 +142,12 @@ class TestChatGPTResponsesAPITransformation: assert "user" not in request assert "temperature" not in request assert "top_p" not in request + assert "context_management" not in request assert "metadata" not in request assert "max_output_tokens" not in request assert "stream_options" not in request assert request["prompt_cache_key"] == "session_123" - assert request["context_management"] == [ - {"type": "compaction", "compact_threshold": 200000} - ] assert request["truncation"] == "auto" assert request["previous_response_id"] == "resp_123" assert request["reasoning"] == {"effort": "medium"}