From 1cd827874f03473c4f91956e3d8bd7b4818afebb Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sun, 10 Aug 2025 09:55:36 -0700 Subject: [PATCH] [Bug Fix] - Allow using `reasoning_effort` for gpt-5 model family and `reasoning` for Responses API (#13475) * test_openai_gpt5_reasoning * test_openai_gpt5_reasoning_effort_parameter * add OpenAIGPT5ResponsesAPIConfig * test_openai_gpt5_reasoning_effort_parameter * fixes --- litellm/utils.py | 5 ++ .../test_openai_responses_api.py | 85 +++++++++++++++++++ tests/llm_translation/test_openai.py | 10 +++ 3 files changed, 100 insertions(+) diff --git a/litellm/utils.py b/litellm/utils.py index 17d10b04b65..1a571d30270 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -6755,6 +6755,11 @@ class ProviderConfigManager: and litellm.openaiOSeriesConfig.is_model_o_series_model(model=model) ): return litellm.openaiOSeriesConfig + elif ( + provider == LlmProviders.OPENAI + and litellm.OpenAIGPT5Config.is_model_gpt_5_model(model=model) + ): + return litellm.OpenAIGPT5Config() elif litellm.LlmProviders.DEEPSEEK == provider: return litellm.DeepSeekChatConfig() elif litellm.LlmProviders.GROQ == provider: diff --git a/tests/llm_responses_api_testing/test_openai_responses_api.py b/tests/llm_responses_api_testing/test_openai_responses_api.py index 427f779cb0a..10cb65d59ff 100644 --- a/tests/llm_responses_api_testing/test_openai_responses_api.py +++ b/tests/llm_responses_api_testing/test_openai_responses_api.py @@ -1398,4 +1398,89 @@ async def test_aresponses_service_tier_and_safety_identifier(): print("Response:", json.dumps(response, indent=4, default=str)) +@pytest.mark.asyncio +async def test_openai_gpt5_reasoning_effort_parameter(): + """Test that reasoning_effort parameter is properly sent in the HTTP request for GPT-5 models.""" + + # Mock response for GPT-5 responses API (correct format) + mock_response = { + "id": "resp_01ABC123", + "object": "response", + "created_at": 1729621667, + "status": "completed", + "model": "gpt-5-mini", + "output": [ + { + "type": "message", + "id": "msg_123", + "status": "completed", + "role": "assistant", + "content": [ + {"type": "output_text", "text": "The capital of France is Paris.", "annotations": []} + ], + } + ], + "parallel_tool_calls": True, + "usage": { + "input_tokens": 15, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens": 8, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 23, + }, + "text": {"format": {"type": "text"}}, + "error": None, + "incomplete_details": None, + "instructions": None, + "metadata": {}, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + "max_output_tokens": None, + "previous_response_id": None, + "reasoning": {"effort": "low", "summary": None}, + "truncation": "disabled", + "user": None, + } + + class MockResponse: + def __init__(self, json_data, status_code): + self._json_data = json_data + self.status_code = status_code + self.text = json.dumps(json_data) + + def json(self): + return self._json_data + + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + new_callable=AsyncMock, + ) as mock_post: + # Configure the mock to return our response + mock_post.return_value = MockResponse(mock_response, 200) + + litellm._turn_on_debug() + litellm.set_verbose = True + + # Call aresponses with reasoning_effort parameter + response = await litellm.aresponses( + model="openai/gpt-5-mini", + input="What is the capital of France?", + reasoning={"effort": "minimal"}, + ) + + # Verify the request was made correctly + mock_post.assert_called_once() + request_body = mock_post.call_args.kwargs["json"] + print("request_body=", json.dumps(request_body, indent=4, default=str)) + print("reasoning=", request_body["reasoning"]) + # Validate that reasoning_effort is present in the request body + assert "reasoning" in request_body, "reasoning should be present in request body" + assert request_body["reasoning"]["effort"] == "minimal", "reasoning_effort should be 'minimal' in request body" + assert request_body["model"] == "gpt-5-mini" + assert request_body["input"] == "What is the capital of France?" + + # Validate the response + print("Response:", json.dumps(response, indent=4, default=str)) diff --git a/tests/llm_translation/test_openai.py b/tests/llm_translation/test_openai.py index f5de082ded2..0121eccaac3 100644 --- a/tests/llm_translation/test_openai.py +++ b/tests/llm_translation/test_openai.py @@ -654,3 +654,13 @@ def test_openai_tool_calling(): } response = litellm.completion(**completion_params) + +@pytest.mark.asyncio +async def test_openai_gpt5_reasoning(): + response = await litellm.acompletion( + model="openai/gpt-5-mini", + messages=[{"role": "user", "content": "What is the capital of France?"}], + reasoning_effort="minimal", + ) + print("response: ", response) + assert response.choices[0].message.content is not None