mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(chatgpt): preserve requested service tier in Responses calls (#40108)
* fix(chatgpt): preserve requested service tier in Responses calls * fix(chatgpt): avoid extra mutable collections in tier filtering * test(chatgpt): refresh service-tier cases after main rebase Keep the tier-preservation regression on the current subscription model and match the moved unit-test suite's formatting. The adapter still forwards preferences without claiming backend priority entitlement. * refactor(chatgpt): map service tier through a lookup table after the allowlist filter --------- Co-authored-by: ryan-crabbe-berri <ryan@berri.ai>
This commit is contained in:
parent
56bba4fbbe
commit
e5873adbc4
2 changed files with 57 additions and 29 deletions
|
|
@ -35,6 +35,8 @@ from ..common_utils import (
|
|||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
_CHATGPT_SERVICE_TIERS: Final = {"default": "default", "priority": "priority", "fast": "priority"}
|
||||
|
||||
|
||||
class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
||||
def __init__(self) -> None:
|
||||
|
|
@ -108,7 +110,11 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
|||
"truncation",
|
||||
}
|
||||
|
||||
return {k: v for k, v in request.items() if k in allowed_keys}
|
||||
filtered: Final = {k: v for k, v in request.items() if k in allowed_keys}
|
||||
service_tier: Final = _CHATGPT_SERVICE_TIERS.get(request.get("service_tier"))
|
||||
if service_tier is not None:
|
||||
filtered["service_tier"] = service_tier
|
||||
return filtered
|
||||
|
||||
def transform_response_api_response(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ Source: litellm/llms/chatgpt/responses/transformation.py
|
|||
|
||||
import json
|
||||
from collections.abc import Generator
|
||||
from typing import Final
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import httpx
|
||||
|
|
@ -30,6 +31,46 @@ def local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> Generator[None, Non
|
|||
|
||||
|
||||
class TestChatGPTResponsesAPITransformation:
|
||||
@pytest.mark.parametrize(
|
||||
("requested_tier", "expected_tier"),
|
||||
[("default", "default"), ("priority", "priority"), ("fast", "priority")],
|
||||
)
|
||||
@pytest.mark.parametrize("effort", ["low", "high"])
|
||||
def test_chatgpt_preserves_service_tier(self, requested_tier: str, expected_tier: str, effort: str) -> None:
|
||||
config: Final = ChatGPTResponsesAPIConfig()
|
||||
request: Final = config.transform_responses_api_request(
|
||||
model="chatgpt/gpt-6.1-sol",
|
||||
input=[{"role": "user", "content": "Reply with OK"}],
|
||||
response_api_optional_request_params={
|
||||
"service_tier": requested_tier,
|
||||
"reasoning": {"effort": effort},
|
||||
"max_output_tokens": 16,
|
||||
"prompt_cache_options": {"ttl": "30m"},
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert request["service_tier"] == expected_tier
|
||||
assert request["reasoning"] == {"effort": effort}
|
||||
assert request["stream"] is True
|
||||
assert request["store"] is False
|
||||
assert "max_output_tokens" not in request
|
||||
assert "prompt_cache_options" not in request
|
||||
|
||||
@pytest.mark.parametrize("requested_tier", [None, "auto", "flex", "unknown"])
|
||||
def test_chatgpt_does_not_introduce_unsupported_service_tier(self, requested_tier: str | None) -> None:
|
||||
config: Final = ChatGPTResponsesAPIConfig()
|
||||
request: Final = config.transform_responses_api_request(
|
||||
model="chatgpt/gpt-6.1-sol",
|
||||
input=[{"role": "user", "content": "Reply with OK"}],
|
||||
response_api_optional_request_params={} if requested_tier is None else {"service_tier": requested_tier},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert "service_tier" not in request
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[
|
||||
|
|
@ -55,7 +96,6 @@ class TestChatGPTResponsesAPITransformation:
|
|||
assert isinstance(config, ChatGPTResponsesAPIConfig)
|
||||
assert config.custom_llm_provider == LlmProviders.CHATGPT
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[
|
||||
|
|
@ -92,14 +132,10 @@ class TestChatGPTResponsesAPITransformation:
|
|||
url = config.get_complete_url(api_base=None, litellm_params={})
|
||||
assert url == "https://chatgpt.example.com/responses"
|
||||
|
||||
custom_url = config.get_complete_url(
|
||||
api_base="https://custom.chatgpt.com", litellm_params={}
|
||||
)
|
||||
custom_url = config.get_complete_url(api_base="https://custom.chatgpt.com", litellm_params={})
|
||||
assert custom_url == "https://custom.chatgpt.com/responses"
|
||||
|
||||
url_with_slash = config.get_complete_url(
|
||||
api_base="https://chatgpt.example.com/", litellm_params={}
|
||||
)
|
||||
url_with_slash = config.get_complete_url(api_base="https://chatgpt.example.com/", litellm_params={})
|
||||
assert url_with_slash == "https://chatgpt.example.com/responses"
|
||||
|
||||
@patch("litellm.llms.chatgpt.responses.transformation.Authenticator")
|
||||
|
|
@ -162,9 +198,7 @@ class TestChatGPTResponsesAPITransformation:
|
|||
"user": "user_123",
|
||||
"temperature": 0.2,
|
||||
"top_p": 0.9,
|
||||
"context_management": [
|
||||
{"type": "compaction", "compact_threshold": 200000}
|
||||
],
|
||||
"context_management": [{"type": "compaction", "compact_threshold": 200000}],
|
||||
"metadata": {"foo": "bar"},
|
||||
"max_output_tokens": 123,
|
||||
"stream_options": {"include_usage": True},
|
||||
|
|
@ -203,9 +237,7 @@ class TestChatGPTResponsesAPITransformation:
|
|||
("chatgpt/gpt-5.3-codex", "gpt-5.3-codex"),
|
||||
],
|
||||
)
|
||||
def test_chatgpt_non_stream_sse_response_parsing(
|
||||
self, model_name: str, response_model: str
|
||||
):
|
||||
def test_chatgpt_non_stream_sse_response_parsing(self, model_name: str, response_model: str):
|
||||
config = ChatGPTResponsesAPIConfig()
|
||||
response_payload = {
|
||||
"id": "resp_test",
|
||||
|
|
@ -228,9 +260,7 @@ class TestChatGPTResponsesAPITransformation:
|
|||
"",
|
||||
]
|
||||
)
|
||||
raw_response = httpx.Response(
|
||||
200, headers={"content-type": "text/event-stream"}, text=sse_body
|
||||
)
|
||||
raw_response = httpx.Response(200, headers={"content-type": "text/event-stream"}, text=sse_body)
|
||||
logging_obj = MagicMock()
|
||||
|
||||
parsed = config.transform_response_api_response(
|
||||
|
|
@ -248,9 +278,7 @@ class TestChatGPTResponsesAPITransformation:
|
|||
("chatgpt/gpt-5.3-codex", "gpt-5.3-codex"),
|
||||
],
|
||||
)
|
||||
def test_chatgpt_non_stream_sse_response_recovers_output_items(
|
||||
self, model_name: str, response_model: str
|
||||
):
|
||||
def test_chatgpt_non_stream_sse_response_recovers_output_items(self, model_name: str, response_model: str):
|
||||
config = ChatGPTResponsesAPIConfig()
|
||||
response_payload = {
|
||||
"id": "resp_test",
|
||||
|
|
@ -273,9 +301,7 @@ class TestChatGPTResponsesAPITransformation:
|
|||
"",
|
||||
]
|
||||
)
|
||||
raw_response = httpx.Response(
|
||||
200, headers={"content-type": "text/event-stream"}, text=sse_body
|
||||
)
|
||||
raw_response = httpx.Response(200, headers={"content-type": "text/event-stream"}, text=sse_body)
|
||||
logging_obj = MagicMock()
|
||||
|
||||
parsed = config.transform_response_api_response(
|
||||
|
|
@ -315,9 +341,7 @@ class TestChatGPTResponsesAPITransformation:
|
|||
"",
|
||||
]
|
||||
)
|
||||
raw_response = httpx.Response(
|
||||
200, headers={"content-type": "text/event-stream"}, text=sse_body
|
||||
)
|
||||
raw_response = httpx.Response(200, headers={"content-type": "text/event-stream"}, text=sse_body)
|
||||
logging_obj = MagicMock()
|
||||
|
||||
parsed = config.transform_response_api_response(
|
||||
|
|
@ -350,9 +374,7 @@ class TestChatGPTResponsesAPITransformation:
|
|||
"",
|
||||
]
|
||||
)
|
||||
raw_response = httpx.Response(
|
||||
502, headers={"content-type": "text/event-stream"}, text=sse_body
|
||||
)
|
||||
raw_response = httpx.Response(502, headers={"content-type": "text/event-stream"}, text=sse_body)
|
||||
logging_obj = MagicMock()
|
||||
|
||||
with pytest.raises(OpenAIError) as exc_info:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue