From 08757c5e7537b44c894be2f844581201230d89d7 Mon Sep 17 00:00:00 2001 From: Steffan Wolter Date: Mon, 7 Sep 2026 12:51:15 +0200 Subject: [PATCH 1/3] fix(chatgpt): preserve requested service tier in Responses calls --- .../llms/chatgpt/responses/transformation.py | 10 ++- .../test_chatgpt_responses_transformation.py | 77 ++++++++++++------- 2 files changed, 59 insertions(+), 28 deletions(-) diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index b96e06be3d8..a649935184a 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -104,7 +104,15 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): "truncation", } - return {k: v for k, v in request.items() if k in allowed_keys} + service_tier: Final[object] = request.get("service_tier") + return { + **{k: v for k, v in request.items() if k in allowed_keys}, + **( + {"service_tier": "priority" if service_tier == "fast" else service_tier} + if service_tier in ("default", "priority", "fast") + else {} + ), + } def transform_response_api_response( self, diff --git a/tests/unit/llms/chatgpt/responses/test_chatgpt_responses_transformation.py b/tests/unit/llms/chatgpt/responses/test_chatgpt_responses_transformation.py index 0b04dd0ed78..f2c2195eaed 100644 --- a/tests/unit/llms/chatgpt/responses/test_chatgpt_responses_transformation.py +++ b/tests/unit/llms/chatgpt/responses/test_chatgpt_responses_transformation.py @@ -6,6 +6,7 @@ Source: litellm/llms/chatgpt/responses/transformation.py import json from collections.abc import Generator +from typing import Final from unittest.mock import MagicMock, patch import httpx @@ -30,6 +31,46 @@ def local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> Generator[None, Non class TestChatGPTResponsesAPITransformation: + @pytest.mark.parametrize( + ("requested_tier", "expected_tier"), + [("default", "default"), ("priority", "priority"), ("fast", "priority")], + ) + @pytest.mark.parametrize("effort", ["low", "high"]) + def test_chatgpt_preserves_service_tier(self, requested_tier: str, expected_tier: str, effort: str) -> None: + config: Final = ChatGPTResponsesAPIConfig() + request: Final = config.transform_responses_api_request( + model="chatgpt/gpt-5.6-sol", + input=[{"role": "user", "content": "Reply with OK"}], + response_api_optional_request_params={ + "service_tier": requested_tier, + "reasoning": {"effort": effort}, + "max_output_tokens": 16, + "prompt_cache_options": {"ttl": "30m"}, + }, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert request["service_tier"] == expected_tier + assert request["reasoning"] == {"effort": effort} + assert request["stream"] is True + assert request["store"] is False + assert "max_output_tokens" not in request + assert "prompt_cache_options" not in request + + @pytest.mark.parametrize("requested_tier", [None, "auto", "flex", "unknown"]) + def test_chatgpt_does_not_introduce_unsupported_service_tier(self, requested_tier: str | None) -> None: + config: Final = ChatGPTResponsesAPIConfig() + request: Final = config.transform_responses_api_request( + model="chatgpt/gpt-5.6-sol", + input=[{"role": "user", "content": "Reply with OK"}], + response_api_optional_request_params={} if requested_tier is None else {"service_tier": requested_tier}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert "service_tier" not in request + @pytest.mark.parametrize( "model_name", [ @@ -92,14 +133,10 @@ class TestChatGPTResponsesAPITransformation: url = config.get_complete_url(api_base=None, litellm_params={}) assert url == "https://chatgpt.example.com/responses" - custom_url = config.get_complete_url( - api_base="https://custom.chatgpt.com", litellm_params={} - ) + custom_url = config.get_complete_url(api_base="https://custom.chatgpt.com", litellm_params={}) assert custom_url == "https://custom.chatgpt.com/responses" - url_with_slash = config.get_complete_url( - api_base="https://chatgpt.example.com/", litellm_params={} - ) + url_with_slash = config.get_complete_url(api_base="https://chatgpt.example.com/", litellm_params={}) assert url_with_slash == "https://chatgpt.example.com/responses" @patch("litellm.llms.chatgpt.responses.transformation.Authenticator") @@ -162,9 +199,7 @@ class TestChatGPTResponsesAPITransformation: "user": "user_123", "temperature": 0.2, "top_p": 0.9, - "context_management": [ - {"type": "compaction", "compact_threshold": 200000} - ], + "context_management": [{"type": "compaction", "compact_threshold": 200000}], "metadata": {"foo": "bar"}, "max_output_tokens": 123, "stream_options": {"include_usage": True}, @@ -203,9 +238,7 @@ class TestChatGPTResponsesAPITransformation: ("chatgpt/gpt-5.3-codex", "gpt-5.3-codex"), ], ) - def test_chatgpt_non_stream_sse_response_parsing( - self, model_name: str, response_model: str - ): + def test_chatgpt_non_stream_sse_response_parsing(self, model_name: str, response_model: str): config = ChatGPTResponsesAPIConfig() response_payload = { "id": "resp_test", @@ -228,9 +261,7 @@ class TestChatGPTResponsesAPITransformation: "", ] ) - raw_response = httpx.Response( - 200, headers={"content-type": "text/event-stream"}, text=sse_body - ) + raw_response = httpx.Response(200, headers={"content-type": "text/event-stream"}, text=sse_body) logging_obj = MagicMock() parsed = config.transform_response_api_response( @@ -248,9 +279,7 @@ class TestChatGPTResponsesAPITransformation: ("chatgpt/gpt-5.3-codex", "gpt-5.3-codex"), ], ) - def test_chatgpt_non_stream_sse_response_recovers_output_items( - self, model_name: str, response_model: str - ): + def test_chatgpt_non_stream_sse_response_recovers_output_items(self, model_name: str, response_model: str): config = ChatGPTResponsesAPIConfig() response_payload = { "id": "resp_test", @@ -273,9 +302,7 @@ class TestChatGPTResponsesAPITransformation: "", ] ) - raw_response = httpx.Response( - 200, headers={"content-type": "text/event-stream"}, text=sse_body - ) + raw_response = httpx.Response(200, headers={"content-type": "text/event-stream"}, text=sse_body) logging_obj = MagicMock() parsed = config.transform_response_api_response( @@ -315,9 +342,7 @@ class TestChatGPTResponsesAPITransformation: "", ] ) - raw_response = httpx.Response( - 200, headers={"content-type": "text/event-stream"}, text=sse_body - ) + raw_response = httpx.Response(200, headers={"content-type": "text/event-stream"}, text=sse_body) logging_obj = MagicMock() parsed = config.transform_response_api_response( @@ -350,9 +375,7 @@ class TestChatGPTResponsesAPITransformation: "", ] ) - raw_response = httpx.Response( - 502, headers={"content-type": "text/event-stream"}, text=sse_body - ) + raw_response = httpx.Response(502, headers={"content-type": "text/event-stream"}, text=sse_body) logging_obj = MagicMock() with pytest.raises(OpenAIError) as exc_info: From 46d9d8dbfb374cf5d82b3fde12e540d4c4b3f41d Mon Sep 17 00:00:00 2001 From: Steffan Wolter Date: Mon, 7 Sep 2026 13:20:32 +0200 Subject: [PATCH 2/3] fix(chatgpt): avoid extra mutable collections in tier filtering --- litellm/llms/chatgpt/responses/transformation.py | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index a649935184a..097e1bd787d 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -104,14 +104,10 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): "truncation", } - service_tier: Final[object] = request.get("service_tier") return { - **{k: v for k, v in request.items() if k in allowed_keys}, - **( - {"service_tier": "priority" if service_tier == "fast" else service_tier} - if service_tier in ("default", "priority", "fast") - else {} - ), + k: "priority" if k == "service_tier" and v == "fast" else v + for k, v in request.items() + if k in allowed_keys or (k == "service_tier" and v in ("default", "priority", "fast")) } def transform_response_api_response( From c70920d10b75002c629378d28692d48e6892f6e1 Mon Sep 17 00:00:00 2001 From: Steffan Wolter Date: Thu, 1 Oct 2026 20:15:14 +0200 Subject: [PATCH 3/3] test(chatgpt): refresh service-tier cases after main rebase Keep the tier-preservation regression on the current subscription model and match the moved unit-test suite's formatting. The adapter still forwards preferences without claiming backend priority entitlement. --- .../responses/test_chatgpt_responses_transformation.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/tests/unit/llms/chatgpt/responses/test_chatgpt_responses_transformation.py b/tests/unit/llms/chatgpt/responses/test_chatgpt_responses_transformation.py index f2c2195eaed..88868844bb1 100644 --- a/tests/unit/llms/chatgpt/responses/test_chatgpt_responses_transformation.py +++ b/tests/unit/llms/chatgpt/responses/test_chatgpt_responses_transformation.py @@ -39,7 +39,7 @@ class TestChatGPTResponsesAPITransformation: def test_chatgpt_preserves_service_tier(self, requested_tier: str, expected_tier: str, effort: str) -> None: config: Final = ChatGPTResponsesAPIConfig() request: Final = config.transform_responses_api_request( - model="chatgpt/gpt-5.6-sol", + model="chatgpt/gpt-6.1-sol", input=[{"role": "user", "content": "Reply with OK"}], response_api_optional_request_params={ "service_tier": requested_tier, @@ -62,7 +62,7 @@ class TestChatGPTResponsesAPITransformation: def test_chatgpt_does_not_introduce_unsupported_service_tier(self, requested_tier: str | None) -> None: config: Final = ChatGPTResponsesAPIConfig() request: Final = config.transform_responses_api_request( - model="chatgpt/gpt-5.6-sol", + model="chatgpt/gpt-6.1-sol", input=[{"role": "user", "content": "Reply with OK"}], response_api_optional_request_params={} if requested_tier is None else {"service_tier": requested_tier}, litellm_params=GenericLiteLLMParams(), @@ -96,7 +96,6 @@ class TestChatGPTResponsesAPITransformation: assert isinstance(config, ChatGPTResponsesAPIConfig) assert config.custom_llm_provider == LlmProviders.CHATGPT - @pytest.mark.parametrize( "model_name", [