From 8e363fe78c29881a20fc61160c90b044d79d4c36 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Wed, 3 Sep 2025 17:58:55 +0530 Subject: [PATCH 1/5] feat: add structured output for sdk --- litellm/responses/main.py | 68 +++++--- litellm/responses/utils.py | 50 +++++- .../base_responses_api.py | 164 ++++++++++++++++-- 3 files changed, 237 insertions(+), 45 deletions(-) diff --git a/litellm/responses/main.py b/litellm/responses/main.py index 9584baf7368..11c88c526ac 100644 --- a/litellm/responses/main.py +++ b/litellm/responses/main.py @@ -1,9 +1,10 @@ import asyncio import contextvars from functools import partial -from typing import Any, Coroutine, Dict, Iterable, List, Literal, Optional, Union +from typing import Any, Coroutine, Dict, Iterable, List, Literal, Optional, Type, Union import httpx +from pydantic import BaseModel import litellm from litellm.constants import request_timeout @@ -135,9 +136,10 @@ async def aresponses_api_with_mcp( ) # Parse MCP tools and separate from other tools - mcp_tools_with_litellm_proxy, other_tools = ( - LiteLLM_Proxy_MCP_Handler._parse_mcp_tools(tools) - ) + ( + mcp_tools_with_litellm_proxy, + other_tools, + ) = LiteLLM_Proxy_MCP_Handler._parse_mcp_tools(tools) # Get available tools from MCP manager if we have MCP tools openai_tools = [] @@ -254,6 +256,7 @@ async def aresponses( stream: Optional[bool] = None, temperature: Optional[float] = None, text: Optional["ResponseText"] = None, + text_format: Optional[Union[Type["BaseModel"], dict]] = None, tool_choice: Optional[ToolChoice] = None, tools: Optional[Iterable[ToolParam]] = None, top_p: Optional[float] = None, @@ -279,6 +282,14 @@ async def aresponses( loop = asyncio.get_event_loop() kwargs["aresponses"] = True + # Convert text_format to text parameter if provided + text = ResponsesAPIRequestUtils.convert_text_format_to_text_param( + text_format=text_format, text=text + ) + if text is not None: + # Update local_vars to include the converted text parameter + local_vars["text"] = text + # get custom llm provider so we can use this for mapping exceptions if custom_llm_provider is None: _, custom_llm_provider, _, _ = litellm.get_llm_provider( @@ -367,6 +378,7 @@ def responses( stream: Optional[bool] = None, temperature: Optional[float] = None, text: Optional["ResponseText"] = None, + text_format: Optional[Union[Type["BaseModel"], dict]] = None, tool_choice: Optional[ToolChoice] = None, tools: Optional[Iterable[ToolParam]] = None, top_p: Optional[float] = None, @@ -399,6 +411,14 @@ def responses( litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None) _is_async = kwargs.pop("aresponses", False) is True + # Convert text_format to text parameter if provided + text = ResponsesAPIRequestUtils.convert_text_format_to_text_param( + text_format=text_format, text=text + ) + if text is not None: + # Update local_vars to include the converted text parameter + local_vars["text"] = text + # get llm provider logic litellm_params = GenericLiteLLMParams(**kwargs) @@ -432,11 +452,11 @@ def responses( ) # get provider config - responses_api_provider_config: Optional[BaseResponsesAPIConfig] = ( - ProviderConfigManager.get_provider_responses_api_config( - model=model, - provider=litellm.LlmProviders(custom_llm_provider), - ) + responses_api_provider_config: Optional[ + BaseResponsesAPIConfig + ] = ProviderConfigManager.get_provider_responses_api_config( + model=model, + provider=litellm.LlmProviders(custom_llm_provider), ) local_vars.update(kwargs) @@ -628,11 +648,11 @@ def delete_responses( raise ValueError("custom_llm_provider is required but passed as None") # get provider config - responses_api_provider_config: Optional[BaseResponsesAPIConfig] = ( - ProviderConfigManager.get_provider_responses_api_config( - model=None, - provider=litellm.LlmProviders(custom_llm_provider), - ) + responses_api_provider_config: Optional[ + BaseResponsesAPIConfig + ] = ProviderConfigManager.get_provider_responses_api_config( + model=None, + provider=litellm.LlmProviders(custom_llm_provider), ) if responses_api_provider_config is None: @@ -807,11 +827,11 @@ def get_responses( raise ValueError("custom_llm_provider is required but passed as None") # get provider config - responses_api_provider_config: Optional[BaseResponsesAPIConfig] = ( - ProviderConfigManager.get_provider_responses_api_config( - model=None, - provider=litellm.LlmProviders(custom_llm_provider), - ) + responses_api_provider_config: Optional[ + BaseResponsesAPIConfig + ] = ProviderConfigManager.get_provider_responses_api_config( + model=None, + provider=litellm.LlmProviders(custom_llm_provider), ) if responses_api_provider_config is None: @@ -963,11 +983,11 @@ def list_input_items( if custom_llm_provider is None: raise ValueError("custom_llm_provider is required but passed as None") - responses_api_provider_config: Optional[BaseResponsesAPIConfig] = ( - ProviderConfigManager.get_provider_responses_api_config( - model=None, - provider=litellm.LlmProviders(custom_llm_provider), - ) + responses_api_provider_config: Optional[ + BaseResponsesAPIConfig + ] = ProviderConfigManager.get_provider_responses_api_config( + model=None, + provider=litellm.LlmProviders(custom_llm_provider), ) if responses_api_provider_config is None: diff --git a/litellm/responses/utils.py b/litellm/responses/utils.py index ac59d28a50d..b66fd0d547e 100644 --- a/litellm/responses/utils.py +++ b/litellm/responses/utils.py @@ -1,5 +1,17 @@ import base64 -from typing import Any, Dict, List, Optional, Union, cast, get_type_hints, overload +from typing import ( + Any, + Dict, + List, + Optional, + Type, + Union, + cast, + get_type_hints, + overload, +) + +from pydantic import BaseModel import litellm from litellm._logging import verbose_logger @@ -8,6 +20,7 @@ from litellm.types.llms.openai import ( ResponseAPIUsage, ResponsesAPIOptionalRequestParams, ResponsesAPIResponse, + ResponseText, ) from litellm.types.responses.main import DecodedResponseId from litellm.types.utils import SpecialEnums, Usage @@ -24,7 +37,6 @@ class ResponsesAPIRequestUtils: custom_llm_provider: Optional[str], model: str, ): - if supported_params is None: return unsupported_params = {} @@ -302,6 +314,40 @@ class ResponsesAPIRequestUtils: ) return decoded_response_id.get("response_id", previous_response_id) + @staticmethod + def convert_text_format_to_text_param( + text_format: Optional[Union[Type["BaseModel"], dict]], + text: Optional["ResponseText"] = None, + ) -> Optional["ResponseText"]: + """ + Convert text_format parameter to text parameter for the responses API. + + Args: + text_format: Pydantic model class or dict to convert to response format + text: Existing text parameter (if provided, text_format is ignored) + + Returns: + ResponseText object with the converted format, or None if conversion fails + """ + if text_format is not None and text is None: + from litellm.llms.base_llm.base_utils import type_to_response_format_param + + # Convert Pydantic model to response format + response_format = type_to_response_format_param(text_format) + if response_format is not None: + # Create ResponseText object with the format + # The responses API expects the format to have name at the top level + text = { + "format": { + "type": response_format["type"], + "name": response_format["json_schema"]["name"], + "schema": response_format["json_schema"]["schema"], + "strict": response_format["json_schema"]["strict"], + } + } + return text + return text + class ResponseAPILoggingUtils: @staticmethod diff --git a/tests/llm_responses_api_testing/base_responses_api.py b/tests/llm_responses_api_testing/base_responses_api.py index 5cb8295b1af..f8fbe53e03d 100644 --- a/tests/llm_responses_api_testing/base_responses_api.py +++ b/tests/llm_responses_api_testing/base_responses_api.py @@ -1,33 +1,25 @@ -import httpx import json -import pytest -import sys -from typing import Any, Dict, List -from unittest.mock import MagicMock, Mock, patch import os -import uuid -import time -import base64 +import sys + +import pytest sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path -import litellm +import json from abc import ABC, abstractmethod -from litellm.integrations.custom_logger import CustomLogger -import json -from litellm.types.utils import StandardLoggingPayload -from litellm.types.llms.openai import ( - ResponseCompletedEvent, - ResponsesAPIResponse, - ResponseAPIUsage, - IncompleteDetails, -) from openai.types.responses.response_create_params import ( ResponseInputParam, ) -from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler + +import litellm +from litellm.types.llms.openai import ( + IncompleteDetails, + ResponseAPIUsage, + ResponsesAPIResponse, +) def validate_responses_api_response(response, final_chunk: bool = False): @@ -537,3 +529,137 @@ class BaseResponsesAPITest(ABC): validate_responses_api_response(final_response, final_chunk=True) assert final_response.output is not None assert len(final_response.output) > 0 + + @pytest.mark.asyncio + async def test_text_format_to_text_conversion(self): + """ + Test that when text_format parameter is passed to litellm.aresponses, + it gets converted to text parameter in the raw API call to OpenAI. + """ + from unittest.mock import AsyncMock, patch + + from pydantic import BaseModel + + class TestResponse(BaseModel): + """Test Pydantic model for structured output""" + + answer: str + confidence: float + + class MockResponse: + """Mock response class for testing""" + + def __init__(self, json_data, status_code): + self._json_data = json_data + self.status_code = status_code + self.text = json.dumps(json_data) + + def json(self): + return self._json_data + + # Mock response from OpenAI + mock_response = { + "id": "resp_123", + "object": "response", + "created_at": 1741476542, + "status": "completed", + "model": "gpt-4o", + "output": [ + { + "type": "message", + "id": "msg_123", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": '{"answer": "Paris", "confidence": 0.95}', + "annotations": [], + } + ], + } + ], + "parallel_tool_calls": True, + "usage": { + "input_tokens": 10, + "output_tokens": 20, + "total_tokens": 30, + "output_tokens_details": {"reasoning_tokens": 0}, + }, + "text": {"format": {"type": "json_object"}}, + "error": None, + "incomplete_details": None, + "instructions": None, + "metadata": {}, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + "max_output_tokens": None, + "previous_response_id": None, + "reasoning": {"effort": None, "summary": None}, + "truncation": "disabled", + "user": None, + } + + base_completion_call_args = self.get_base_completion_call_args() + + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + new_callable=AsyncMock, + ) as mock_post: + # Configure the mock to return our response + mock_post.return_value = MockResponse(mock_response, 200) + + litellm._turn_on_debug() + litellm.set_verbose = True + + # Call aresponses with text_format parameter + response = await litellm.aresponses( + input="What is the capital of France?", + text_format=TestResponse, + **base_completion_call_args, + ) + + # Verify the request was made correctly + mock_post.assert_called_once() + request_body = mock_post.call_args.kwargs["json"] + print("Request body:", json.dumps(request_body, indent=4)) + + # Validate that text_format was converted to text parameter + assert ( + "text" in request_body + ), "text parameter should be present in request body" + assert ( + "text_format" not in request_body + ), "text_format should not be in request body" + + # Validate the text parameter structure + text_param = request_body["text"] + assert "format" in text_param, "text parameter should have format field" + assert ( + text_param["format"]["type"] == "json_schema" + ), "format type should be json_schema" + assert "name" in text_param["format"], "format should have name field" + assert ( + text_param["format"]["name"] == "TestResponse" + ), "format name should match Pydantic model name" + assert "schema" in text_param["format"], "format should have schema field" + assert "strict" in text_param["format"], "format should have strict field" + + # Validate the schema structure + schema = text_param["format"]["schema"] + assert schema["type"] == "object", "schema type should be object" + assert "properties" in schema, "schema should have properties" + assert ( + "answer" in schema["properties"] + ), "schema should have answer property" + assert ( + "confidence" in schema["properties"] + ), "schema should have confidence property" + + # Validate other request parameters + assert request_body["input"] == "What is the capital of France?" + + # Validate the response + print("Response:", json.dumps(response, indent=4, default=str)) From f7f106f8b8c5ca90a1c9c67457edb08a2f24f1e4 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 5 Sep 2025 10:00:30 +0530 Subject: [PATCH 2/5] Move test to test_litellm/ folder --- .../responses/test_text_format_conversion.py | 161 ++++++++++++++++++ 1 file changed, 161 insertions(+) create mode 100644 tests/test_litellm/responses/test_text_format_conversion.py diff --git a/tests/test_litellm/responses/test_text_format_conversion.py b/tests/test_litellm/responses/test_text_format_conversion.py new file mode 100644 index 00000000000..20a87a4abbb --- /dev/null +++ b/tests/test_litellm/responses/test_text_format_conversion.py @@ -0,0 +1,161 @@ +import json +import os +import sys + +import pytest +from pydantic import BaseModel + +sys.path.insert( + 0, os.path.abspath("../../..") +) # Adds the parent directory to the system path + +import litellm +from litellm.types.llms.openai import ( + IncompleteDetails, + ResponseAPIUsage, + ResponsesAPIResponse, +) + + +class TestTextFormatConversion: + """Test text_format to text parameter conversion for responses API""" + + def get_base_completion_call_args(self): + """Get base arguments for completion call""" + return { + "model": "gpt-4o", + "api_key": "test-key", + "api_base": "https://api.openai.com/v1", + } + + @pytest.mark.asyncio + async def test_text_format_to_text_conversion(self): + """ + Test that when text_format parameter is passed to litellm.aresponses, + it gets converted to text parameter in the raw API call to OpenAI. + """ + from unittest.mock import AsyncMock, patch + + class TestResponse(BaseModel): + """Test Pydantic model for structured output""" + + answer: str + confidence: float + + class MockResponse: + """Mock response class for testing""" + + def __init__(self, json_data, status_code): + self._json_data = json_data + self.status_code = status_code + self.text = json.dumps(json_data) + + def json(self): + return self._json_data + + # Mock response from OpenAI + mock_response = { + "id": "resp_123", + "object": "response", + "created_at": 1741476542, + "status": "completed", + "model": "gpt-4o", + "output": [ + { + "type": "message", + "id": "msg_123", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": '{"answer": "Paris", "confidence": 0.95}', + "annotations": [], + } + ], + } + ], + "parallel_tool_calls": True, + "usage": { + "input_tokens": 10, + "output_tokens": 20, + "total_tokens": 30, + "output_tokens_details": {"reasoning_tokens": 0}, + }, + "text": {"format": {"type": "json_object"}}, + "error": None, + "incomplete_details": None, + "instructions": None, + "metadata": {}, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + "max_output_tokens": None, + "previous_response_id": None, + "reasoning": {"effort": None, "summary": None}, + "truncation": "disabled", + "user": None, + } + + base_completion_call_args = self.get_base_completion_call_args() + + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + new_callable=AsyncMock, + ) as mock_post: + # Configure the mock to return our response + mock_post.return_value = MockResponse(mock_response, 200) + + litellm._turn_on_debug() + litellm.set_verbose = True + + # Call aresponses with text_format parameter + response = await litellm.aresponses( + input="What is the capital of France?", + text_format=TestResponse, + **base_completion_call_args, + ) + + # Verify the request was made correctly + mock_post.assert_called_once() + request_body = mock_post.call_args.kwargs["json"] + print("Request body:", json.dumps(request_body, indent=4)) + + # Validate that text_format was converted to text parameter + assert ( + "text" in request_body + ), "text parameter should be present in request body" + assert ( + "text_format" not in request_body + ), "text_format should not be in request body" + + # Validate the text parameter structure + text_param = request_body["text"] + assert "format" in text_param, "text parameter should have format field" + assert ( + text_param["format"]["type"] == "json_schema" + ), "format type should be json_schema" + assert "name" in text_param["format"], "format should have name field" + assert ( + text_param["format"]["name"] == "TestResponse" + ), "format name should match Pydantic model name" + assert "schema" in text_param["format"], "format should have schema field" + assert "strict" in text_param["format"], "format should have strict field" + + # Validate the schema structure + schema = text_param["format"]["schema"] + assert schema["type"] == "object", "schema type should be object" + assert "properties" in schema, "schema should have properties" + assert ( + "answer" in schema["properties"] + ), "schema should have answer property" + assert ( + "confidence" in schema["properties"] + ), "schema should have confidence property" + + # Validate other request parameters + assert request_body["input"] == "What is the capital of France?" + + # Validate the response + print("Response:", json.dumps(response, indent=4, default=str)) From ad9f54a192a9a53ac38dedf1faa5a6467921a682 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 5 Sep 2025 10:08:34 +0530 Subject: [PATCH 3/5] Move test to test_litellm/ folder --- .../base_responses_api.py | 165 +---- .../responses/test_text_format_conversion.py | 594 +++++++++++++++++- 2 files changed, 602 insertions(+), 157 deletions(-) diff --git a/tests/llm_responses_api_testing/base_responses_api.py b/tests/llm_responses_api_testing/base_responses_api.py index 2939e884a56..fc6983520fd 100644 --- a/tests/llm_responses_api_testing/base_responses_api.py +++ b/tests/llm_responses_api_testing/base_responses_api.py @@ -1,25 +1,33 @@ +import httpx import json -import os -import sys - import pytest +import sys +from typing import Any, Dict, List +from unittest.mock import MagicMock, Mock, patch +import os +import uuid +import time +import base64 sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path -import json +import litellm from abc import ABC, abstractmethod +from litellm.integrations.custom_logger import CustomLogger +import json +from litellm.types.utils import StandardLoggingPayload +from litellm.types.llms.openai import ( + ResponseCompletedEvent, + ResponsesAPIResponse, + ResponseAPIUsage, + IncompleteDetails, +) from openai.types.responses.response_create_params import ( ResponseInputParam, ) - -import litellm -from litellm.types.llms.openai import ( - IncompleteDetails, - ResponseAPIUsage, - ResponsesAPIResponse, -) +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler def validate_responses_api_response(response, final_chunk: bool = False): @@ -529,140 +537,6 @@ class BaseResponsesAPITest(ABC): validate_responses_api_response(final_response, final_chunk=True) assert final_response.output is not None - @pytest.mark.asyncio - async def test_text_format_to_text_conversion(self): - """ - Test that when text_format parameter is passed to litellm.aresponses, - it gets converted to text parameter in the raw API call to OpenAI. - """ - from unittest.mock import AsyncMock, patch - - from pydantic import BaseModel - - class TestResponse(BaseModel): - """Test Pydantic model for structured output""" - - answer: str - confidence: float - - class MockResponse: - """Mock response class for testing""" - - def __init__(self, json_data, status_code): - self._json_data = json_data - self.status_code = status_code - self.text = json.dumps(json_data) - - def json(self): - return self._json_data - - # Mock response from OpenAI - mock_response = { - "id": "resp_123", - "object": "response", - "created_at": 1741476542, - "status": "completed", - "model": "gpt-4o", - "output": [ - { - "type": "message", - "id": "msg_123", - "status": "completed", - "role": "assistant", - "content": [ - { - "type": "output_text", - "text": '{"answer": "Paris", "confidence": 0.95}', - "annotations": [], - } - ], - } - ], - "parallel_tool_calls": True, - "usage": { - "input_tokens": 10, - "output_tokens": 20, - "total_tokens": 30, - "output_tokens_details": {"reasoning_tokens": 0}, - }, - "text": {"format": {"type": "json_object"}}, - "error": None, - "incomplete_details": None, - "instructions": None, - "metadata": {}, - "temperature": 1.0, - "tool_choice": "auto", - "tools": [], - "top_p": 1.0, - "max_output_tokens": None, - "previous_response_id": None, - "reasoning": {"effort": None, "summary": None}, - "truncation": "disabled", - "user": None, - } - - base_completion_call_args = self.get_base_completion_call_args() - - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - new_callable=AsyncMock, - ) as mock_post: - # Configure the mock to return our response - mock_post.return_value = MockResponse(mock_response, 200) - - litellm._turn_on_debug() - litellm.set_verbose = True - - # Call aresponses with text_format parameter - response = await litellm.aresponses( - input="What is the capital of France?", - text_format=TestResponse, - **base_completion_call_args, - ) - - # Verify the request was made correctly - mock_post.assert_called_once() - request_body = mock_post.call_args.kwargs["json"] - print("Request body:", json.dumps(request_body, indent=4)) - - # Validate that text_format was converted to text parameter - assert ( - "text" in request_body - ), "text parameter should be present in request body" - assert ( - "text_format" not in request_body - ), "text_format should not be in request body" - - # Validate the text parameter structure - text_param = request_body["text"] - assert "format" in text_param, "text parameter should have format field" - assert ( - text_param["format"]["type"] == "json_schema" - ), "format type should be json_schema" - assert "name" in text_param["format"], "format should have name field" - assert ( - text_param["format"]["name"] == "TestResponse" - ), "format name should match Pydantic model name" - assert "schema" in text_param["format"], "format should have schema field" - assert "strict" in text_param["format"], "format should have strict field" - - # Validate the schema structure - schema = text_param["format"]["schema"] - assert schema["type"] == "object", "schema type should be object" - assert "properties" in schema, "schema should have properties" - assert ( - "answer" in schema["properties"] - ), "schema should have answer property" - assert ( - "confidence" in schema["properties"] - ), "schema should have confidence property" - - # Validate other request parameters - assert request_body["input"] == "What is the capital of France?" - - # Validate the response - print("Response:", json.dumps(response, indent=4, default=str)) - def test_openai_responses_api_dict_input_filtering(self): """ Test that regular dict inputs with status fields are properly filtered @@ -716,4 +590,3 @@ class BaseResponsesAPITest(ABC): assert function_call_item["status"] == "completed", "status value should be preserved" print("✅ OpenAI Responses API dict input filtering test passed") - diff --git a/tests/test_litellm/responses/test_text_format_conversion.py b/tests/test_litellm/responses/test_text_format_conversion.py index 20a87a4abbb..dafccfaf2b9 100644 --- a/tests/test_litellm/responses/test_text_format_conversion.py +++ b/tests/test_litellm/responses/test_text_format_conversion.py @@ -3,11 +3,16 @@ import os import sys import pytest -from pydantic import BaseModel sys.path.insert( - 0, os.path.abspath("../../..") + 0, os.path.abspath("../..") ) # Adds the parent directory to the system path +import json +from abc import ABC, abstractmethod + +from openai.types.responses.response_create_params import ( + ResponseInputParam, +) import litellm from litellm.types.llms.openai import ( @@ -17,16 +22,512 @@ from litellm.types.llms.openai import ( ) -class TestTextFormatConversion: - """Test text_format to text parameter conversion for responses API""" +def validate_responses_api_response(response, final_chunk: bool = False): + """ + Validate that a response from litellm.responses() or litellm.aresponses() + conforms to the expected ResponsesAPIResponse structure. - def get_base_completion_call_args(self): - """Get base arguments for completion call""" - return { - "model": "gpt-4o", - "api_key": "test-key", - "api_base": "https://api.openai.com/v1", - } + Args: + response: The response object to validate + + Raises: + AssertionError: If the response doesn't match the expected structure + """ + # Validate response structure + print("response=", json.dumps(response, indent=4, default=str)) + assert isinstance( + response, ResponsesAPIResponse + ), "Response should be an instance of ResponsesAPIResponse" + + # Required fields + assert "id" in response and isinstance( + response["id"], str + ), "Response should have a string 'id' field" + assert "created_at" in response and isinstance( + response["created_at"], int + ), "Response should have an integer 'created_at' field" + assert "output" in response and isinstance( + response["output"], list + ), "Response should have a list 'output' field" + assert "parallel_tool_calls" in response and isinstance( + response["parallel_tool_calls"], bool + ), "Response should have a boolean 'parallel_tool_calls' field" + + # Optional fields with their expected types + optional_fields = { + "error": (dict, type(None)), # error can be dict or None + "incomplete_details": (IncompleteDetails, type(None)), + "instructions": (str, type(None)), + "metadata": dict, + "model": str, + "object": str, + "temperature": (int, float, type(None)), + "tool_choice": (dict, str), + "tools": list, + "top_p": (int, float, type(None)), + "max_output_tokens": (int, type(None)), + "previous_response_id": (str, type(None)), + "reasoning": dict, + "status": str, + "text": dict, + "truncation": (str, type(None)), + "usage": ResponseAPIUsage, + "user": (str, type(None)), + "store": (bool, type(None)), + } + if final_chunk is False: + optional_fields["usage"] = type(None) + + for field, expected_type in optional_fields.items(): + if field in response: + assert isinstance( + response[field], expected_type + ), f"Field '{field}' should be of type {expected_type}, but got {type(response[field])}" + + # Check if output has at least one item + if final_chunk is True: + assert ( + len(response["output"]) > 0 + ), "Response 'output' field should have at least one item" + + return True # Return True if validation passes + + +class BaseResponsesAPITest(ABC): + """ + Abstract base test class that enforces a common test across all test classes. + """ + + @abstractmethod + def get_base_completion_call_args(self) -> dict: + """Must return the base completion call args""" + pass + + def get_base_completion_reasoning_call_args(self) -> dict: + """Must return the base completion reasoning call args""" + return None + + @pytest.mark.parametrize("sync_mode", [True, False]) + @pytest.mark.asyncio + async def test_basic_openai_responses_api(self, sync_mode): + litellm._turn_on_debug() + litellm.set_verbose = True + base_completion_call_args = self.get_base_completion_call_args() + try: + if sync_mode: + response = litellm.responses( + input="Basic ping", + max_output_tokens=20, + **base_completion_call_args, + ) + else: + response = await litellm.aresponses( + input="Basic ping", + max_output_tokens=20, + **base_completion_call_args, + ) + except litellm.InternalServerError: + pytest.skip("Skipping test due to litellm.InternalServerError") + print("litellm response=", json.dumps(response, indent=4, default=str)) + + # Use the helper function to validate the response + validate_responses_api_response(response, final_chunk=True) + + @pytest.mark.parametrize("sync_mode", [True, False]) + @pytest.mark.asyncio + @pytest.mark.flaky(retries=3, delay=2) + async def test_basic_openai_responses_api_streaming(self, sync_mode): + litellm._turn_on_debug() + base_completion_call_args = self.get_base_completion_call_args() + collected_content_string = "" + response_completed_event = None + if sync_mode: + response = litellm.responses( + input="Basic ping", stream=True, **base_completion_call_args + ) + for event in response: + print("litellm response=", json.dumps(event, indent=4, default=str)) + if event.type == "response.output_text.delta": + collected_content_string += event.delta + elif event.type == "response.completed": + response_completed_event = event + else: + response = await litellm.aresponses( + input="Basic ping", stream=True, **base_completion_call_args + ) + async for event in response: + print("litellm response=", json.dumps(event, indent=4, default=str)) + if event.type == "response.output_text.delta": + collected_content_string += event.delta + elif event.type == "response.completed": + response_completed_event = event + + # assert the delta chunks content had len(collected_content_string) > 0 + # this content is typically rendered on chat ui's + assert len(collected_content_string) > 0 + + # assert the response completed event is not None + assert response_completed_event is not None + + # assert the response completed event has a response + assert response_completed_event.response is not None + + # assert the response completed event includes the usage + assert response_completed_event.response.usage is not None + + # basic test assert the usage seems reasonable + print( + "response_completed_event.response.usage=", + response_completed_event.response.usage, + ) + assert ( + response_completed_event.response.usage.input_tokens > 0 + and response_completed_event.response.usage.input_tokens < 100 + ) + assert ( + response_completed_event.response.usage.output_tokens > 0 + and response_completed_event.response.usage.output_tokens < 2000 + ) + assert ( + response_completed_event.response.usage.total_tokens > 0 + and response_completed_event.response.usage.total_tokens < 2000 + ) + + # total tokens should be the sum of input and output tokens + assert ( + response_completed_event.response.usage.total_tokens + == response_completed_event.response.usage.input_tokens + + response_completed_event.response.usage.output_tokens + ) + + @pytest.mark.parametrize("sync_mode", [False, True]) + @pytest.mark.asyncio + async def test_basic_openai_responses_delete_endpoint(self, sync_mode): + litellm._turn_on_debug() + litellm.set_verbose = True + base_completion_call_args = self.get_base_completion_call_args() + if sync_mode: + response = litellm.responses( + input="Basic ping", max_output_tokens=20, **base_completion_call_args + ) + + # delete the response + if isinstance(response, ResponsesAPIResponse): + litellm.delete_responses( + response_id=response.id, **base_completion_call_args + ) + else: + raise ValueError("response is not a ResponsesAPIResponse") + else: + response = await litellm.aresponses( + input="Basic ping", max_output_tokens=20, **base_completion_call_args + ) + + # async delete the response + if isinstance(response, ResponsesAPIResponse): + await litellm.adelete_responses( + response_id=response.id, **base_completion_call_args + ) + else: + raise ValueError("response is not a ResponsesAPIResponse") + + @pytest.mark.parametrize("sync_mode", [True, False]) + @pytest.mark.flaky(retries=3, delay=2) + @pytest.mark.asyncio + async def test_basic_openai_responses_streaming_delete_endpoint(self, sync_mode): + # litellm._turn_on_debug() + # litellm.set_verbose = True + base_completion_call_args = self.get_base_completion_call_args() + response_id = None + if sync_mode: + response_id = None + response = litellm.responses( + input="Basic ping", + max_output_tokens=20, + stream=True, + **base_completion_call_args, + ) + for event in response: + print("litellm response=", json.dumps(event, indent=4, default=str)) + if "response" in event: + response_obj = event.get("response") + if response_obj is not None: + response_id = response_obj.get("id") + print("got response_id=", response_id) + + # delete the response + assert response_id is not None + litellm.delete_responses( + response_id=response_id, **base_completion_call_args + ) + else: + response = await litellm.aresponses( + input="Basic ping", + max_output_tokens=20, + stream=True, + **base_completion_call_args, + ) + async for event in response: + print("litellm response=", json.dumps(event, indent=4, default=str)) + if "response" in event: + response_obj = event.get("response") + if response_obj is not None: + response_id = response_obj.get("id") + print("got response_id=", response_id) + + # delete the response + assert response_id is not None + await litellm.adelete_responses( + response_id=response_id, **base_completion_call_args + ) + + @pytest.mark.parametrize("sync_mode", [False, True]) + @pytest.mark.flaky(retries=3, delay=2) + @pytest.mark.asyncio + async def test_basic_openai_responses_get_endpoint(self, sync_mode): + litellm._turn_on_debug() + litellm.set_verbose = True + base_completion_call_args = self.get_base_completion_call_args() + if sync_mode: + response = litellm.responses( + input="Basic ping", max_output_tokens=20, **base_completion_call_args + ) + + # get the response + if isinstance(response, ResponsesAPIResponse): + result = litellm.get_responses( + response_id=response.id, **base_completion_call_args + ) + assert result is not None + assert result.id == response.id + assert result.output == response.output + else: + raise ValueError("response is not a ResponsesAPIResponse") + else: + response = await litellm.aresponses( + input="Basic ping", max_output_tokens=20, **base_completion_call_args + ) + # async get the response + if isinstance(response, ResponsesAPIResponse): + result = await litellm.aget_responses( + response_id=response.id, **base_completion_call_args + ) + assert result is not None + assert result.id == response.id + assert result.output == response.output + else: + raise ValueError("response is not a ResponsesAPIResponse") + + @pytest.mark.asyncio + @pytest.mark.flaky(retries=3, delay=2) + async def test_basic_openai_list_input_items_endpoint(self): + """Test that calls the OpenAI List Input Items endpoint""" + litellm._turn_on_debug() + + response = await litellm.aresponses( + model="gpt-4o", + input="Tell me a three sentence bedtime story about a unicorn.", + ) + print("Initial response=", json.dumps(response, indent=4, default=str)) + + response_id = response.get("id") + assert response_id is not None, "Response should have an ID" + print(f"Got response_id: {response_id}") + + list_items_response = await litellm.alist_input_items( + response_id=response_id, + limit=20, + order="desc", + ) + print( + "List items response=", + json.dumps(list_items_response, indent=4, default=str), + ) + + @pytest.mark.asyncio + async def test_multiturn_responses_api(self): + litellm._turn_on_debug() + litellm.set_verbose = True + try: + base_completion_call_args = self.get_base_completion_call_args() + response_1 = await litellm.aresponses( + input="Basic ping", max_output_tokens=20, **base_completion_call_args + ) + + # follow up with a second request + response_1_id = response_1.id + response_2 = await litellm.aresponses( + input="Basic ping", + max_output_tokens=20, + previous_response_id=response_1_id, + **base_completion_call_args, + ) + + # assert the response is not None + assert response_1 is not None + assert response_2 is not None + except litellm.InternalServerError: + pytest.skip("Skipping test due to litellm.InternalServerError") + + @pytest.mark.asyncio + async def test_responses_api_with_tool_calls(self): + """Test that calls the Responses API with tool calls including function call and output""" + litellm._turn_on_debug() + litellm.set_verbose = True + base_completion_call_args = self.get_base_completion_call_args() + + # Define the input with message, function call, and function call output + input_data: ResponseInputParam = [ + { + "type": "message", + "role": "user", + "content": "How is the weather in São Paulo today ?", + }, + { + "type": "function_call", + "arguments": '{"location": "São Paulo, Brazil"}', + "call_id": "fc_1fe70e2a-a596-45ef-b72c-9b8567c460e5", + "name": "get_weather", + "id": "fc_1fe70e2a-a596-45ef-b72c-9b8567c460e5", + "status": "completed", + }, + { + "type": "function_call_output", + "call_id": "fc_1fe70e2a-a596-45ef-b72c-9b8567c460e5", + "output": "Rainy", + }, + ] + + # Define the tools + tools = [ + { + "type": "function", + "name": "get_weather", + "description": "Get current temperature for a given location.", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "City and country e.g. Bogotá, Colombia", + } + }, + "required": ["location"], + "additionalProperties": False, + }, + } + ] + + try: + # Make the responses API call + response = await litellm.aresponses( + input=input_data, store=False, tools=tools, **base_completion_call_args + ) + except litellm.InternalServerError: + pytest.skip("Skipping test due to litellm.InternalServerError") + + print("litellm response=", json.dumps(response, indent=4, default=str)) + + # Validate the response structure + validate_responses_api_response(response, final_chunk=True) + + # Additional assertions specific to tool calls + assert response is not None + assert "output" in response + assert len(response["output"]) > 0 + + @pytest.mark.asyncio + async def test_responses_api_multi_turn_with_reasoning_and_structured_output(self): + """ + Test multi-turn conversation with reasoning, structured output, and tool calls. + + This test validates: + - First call: Model uses reasoning to process a question and makes a tool call + - Tool call handling: Function call output is properly processed + - Second call: Model produces structured output incorporating tool results + - Structured output: Response conforms to defined Pydantic model schema + """ + from pydantic import BaseModel + + litellm._turn_on_debug() + litellm.set_verbose = True + base_completion_call_args = self.get_base_completion_reasoning_call_args() + if base_completion_call_args is None: + pytest.skip("Skipping test due to no base completion reasoning call args") + + # Define tools for the conversation + tools = [{"type": "function", "name": "get_today"}] + + # Define structured output schema + class Output(BaseModel): + today: str + number_of_r: str + + # Initial conversation input + input_messages = [ + { + "role": "user", + "content": "How many r in strrawberrry? While you're thinking, you should call tool get_today. Then you output the today and number of r", + } + ] + + # First call - should trigger reasoning and tool call + response = await litellm.aresponses( + input=input_messages, + tools=tools, + reasoning={"effort": "low", "summary": "detailed"}, + text_format=Output, + **base_completion_call_args, + ) + + print("First call output:") + print(json.dumps(response.output, indent=4, default=str)) + + # Validate first response structure + validate_responses_api_response(response, final_chunk=True) + assert response.output is not None + assert len(response.output) > 0 + + # Extend input with first response output + input_messages.extend(response.output) + + # Process any tool calls and add function outputs + function_outputs = [] + for item in response.output: + if hasattr(item, "type") and item.type in [ + "function_call", + "custom_tool_call", + ]: + if hasattr(item, "name") and item.name == "get_today": + function_outputs.append( + { + "type": "function_call_output", + "call_id": item.call_id, + "output": "2025-01-15", + } + ) + + # Add function outputs to conversation + input_messages.extend(function_outputs) + + print("Second call input:") + print(json.dumps(input_messages, indent=4, default=str)) + + # Second call - should produce structured output + final_response = await litellm.aresponses( + input=input_messages, + tools=tools, + reasoning={"effort": "low", "summary": "detailed"}, + text_format=Output, + **base_completion_call_args, + ) + + print("Second call output:") + print(json.dumps(final_response.output, indent=4, default=str)) + + # Validate final response structure + validate_responses_api_response(final_response, final_chunk=True) + assert final_response.output is not None @pytest.mark.asyncio async def test_text_format_to_text_conversion(self): @@ -36,6 +537,8 @@ class TestTextFormatConversion: """ from unittest.mock import AsyncMock, patch + from pydantic import BaseModel + class TestResponse(BaseModel): """Test Pydantic model for structured output""" @@ -159,3 +662,72 @@ class TestTextFormatConversion: # Validate the response print("Response:", json.dumps(response, indent=4, default=str)) + + def test_openai_responses_api_dict_input_filtering(self): + """ + Test that regular dict inputs with status fields are properly filtered + to replicate exclude_unset=True behavior for non-Pydantic objects. + """ + from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig + + # Test input with regular dict objects (like from JSON) + test_input = [ + { + "role": "user", + "content": "test" + }, + { + "id": "rs_123", + "summary": [{"text": "test", "type": "summary_text"}], + "type": "reasoning", + "content": None, # Should be filtered out + "encrypted_content": None, # Should be filtered out + "status": None # Should be filtered out + }, + { + "arguments": "{}", + "call_id": "call_123", + "name": "get_today", + "type": "function_call", + "id": "fc_123", + "status": "completed" # Should be preserved (not a default field) + } + ] + + config = OpenAIResponsesAPIConfig() + validated_input = config._validate_input_param(test_input) + + # Verify the results + assert len(validated_input) == 3 + + # Check reasoning item (index 1) + reasoning_item = validated_input[1] + assert reasoning_item["type"] == "reasoning" + assert "status" not in reasoning_item, "status field should be filtered out from reasoning item" + assert "content" not in reasoning_item, "content field should be filtered out from reasoning item" + assert "encrypted_content" not in reasoning_item, "encrypted_content field should be filtered out from reasoning item" + assert "id" in reasoning_item, "id field should be preserved" + assert "summary" in reasoning_item, "summary field should be preserved" + + # Check function call item (index 2) + function_call_item = validated_input[2] + assert function_call_item["type"] == "function_call" + assert "status" in function_call_item, "status field should be preserved in function call item" + assert function_call_item["status"] == "completed", "status value should be preserved" + + print("✅ OpenAI Responses API dict input filtering test passed") + + +class TestOpenAIResponsesAPITest(BaseResponsesAPITest): + """Concrete test class for OpenAI Responses API tests""" + + def get_base_completion_call_args(self): + return { + "model": "openai/gpt-4o", + } + + def get_base_completion_reasoning_call_args(self): + return { + "model": "openai/gpt-5-mini", + } + From 4fefac1bf2ec414b73c5403ce9d35935fbd81410 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 5 Sep 2025 10:24:55 +0530 Subject: [PATCH 4/5] Move test to test_litellm/ folder --- .../test_openai_responses_transformation.py | 247 +----------------- 1 file changed, 1 insertion(+), 246 deletions(-) diff --git a/tests/test_litellm/llms/openai/responses/test_openai_responses_transformation.py b/tests/test_litellm/llms/openai/responses/test_openai_responses_transformation.py index 21232161d0c..a6a34518098 100644 --- a/tests/test_litellm/llms/openai/responses/test_openai_responses_transformation.py +++ b/tests/test_litellm/llms/openai/responses/test_openai_responses_transformation.py @@ -667,249 +667,4 @@ def test_get_supported_openai_params(): assert "temperature" in params assert "stream" in params assert "background" in params - assert "stream" in params - - -class TestOpenAIFieldExclusionRegistry: - """Test suite for the OpenAI Field Exclusion Registry system""" - - def setup_method(self): - """Setup test fixtures""" - from litellm.llms.openai.responses.transformation import ( - OpenAIFieldExclusionRegistry, - OpenAIResponsesAPIConfig - ) - self.registry = OpenAIFieldExclusionRegistry - self.config = OpenAIResponsesAPIConfig() - - def test_registry_initialization(self): - """Test that the registry is properly initialized with ResponseReasoningItem""" - # Test that we can get excluded fields (should not be empty if ResponseReasoningItem is registered) - all_excluded_fields = self.registry.get_all_excluded_fields() - - # The registry should have at least some fields if ResponseReasoningItem was successfully registered - # If OpenAI SDK is not available, this might be empty, which is also valid - assert isinstance(all_excluded_fields, set), "get_all_excluded_fields should return a set" - - # If we have the OpenAI SDK available, we should have the expected fields - try: - from openai.types.responses import ResponseReasoningItem - reasoning_fields = self.registry.get_excluded_fields_for_model(ResponseReasoningItem) - expected_fields = {'status', 'content', 'encrypted_content'} - assert expected_fields.issubset(reasoning_fields), f"Expected fields {expected_fields} to be subset of {reasoning_fields}" - except ImportError: - # If OpenAI SDK is not available, that's fine - the registry should handle this gracefully - pytest.skip("OpenAI SDK not available, skipping ResponseReasoningItem specific tests") - - def test_register_model_functionality(self): - """Test that we can register new models to the registry""" - from pydantic import BaseModel - from typing import Optional - - # Create a test model with default None fields - class TestResponseModel(BaseModel): - id: str - type: str = "test" - status: Optional[str] = None - content: Optional[str] = None - required_field: str - - # Register the test model - self.registry.register_model(TestResponseModel) - - # Verify it was registered and fields are detected - excluded_fields = self.registry.get_excluded_fields_for_model(TestResponseModel) - expected_excluded = {'status', 'content'} # Fields with default None - - assert expected_excluded.issubset(excluded_fields), f"Expected {expected_excluded} to be in {excluded_fields}" - assert 'id' not in excluded_fields, "Required field 'id' should not be excluded" - assert 'required_field' not in excluded_fields, "Required field 'required_field' should not be excluded" - - def test_get_all_excluded_fields(self): - """Test that get_all_excluded_fields aggregates fields from all registered models""" - all_fields_before = self.registry.get_all_excluded_fields() - - # Create and register a test model - from pydantic import BaseModel - from typing import Optional - - class AnotherTestModel(BaseModel): - id: str - unique_field: Optional[str] = None - - self.registry.register_model(AnotherTestModel) - - all_fields_after = self.registry.get_all_excluded_fields() - - # The new fields should be included - assert 'unique_field' in all_fields_after, "New model's excluded field should be included" - assert len(all_fields_after) >= len(all_fields_before), "Should have at least as many fields as before" - - def test_convenience_registration_method(self): - """Test the convenience method for registering models""" - from pydantic import BaseModel - from typing import Optional - - class ConvenienceTestModel(BaseModel): - id: str - convenience_field: Optional[str] = None - - # Use the convenience method - self.config.register_model_for_field_exclusion(ConvenienceTestModel) - - # Verify it was registered - excluded_fields = self.registry.get_excluded_fields_for_model(ConvenienceTestModel) - assert 'convenience_field' in excluded_fields, "Field should be excluded after registration" - - def test_field_filtering_with_registry(self): - """Test that the field filtering works correctly with the registry""" - - # Test data that matches the structure of ResponseReasoningItem - test_input = [ - { - "role": "user", - "content": "test message" - }, - { - "id": "reasoning-123", - "type": "reasoning", - "status": None, # Should be filtered out - "content": None, # Should be filtered out - "encrypted_content": None, # Should be filtered out - "summary": [{"text": "This reasoning shows...", "type": "summary_text"}], - "role": "assistant" - }, - { - "id": "message-456", - "type": "message", - "status": "completed", # Should be preserved (not None) - "content": "Hello! How can I help?", # Should be preserved (not None) - "role": "assistant" - } - ] - - # Process the input through the validation - result = self.config._validate_input_param(test_input) - - # Verify the structure - assert len(result) == 3, "Should have 3 items" - - # Check the reasoning item (index 1) - reasoning_item = result[1] - assert reasoning_item["type"] == "reasoning" - assert reasoning_item["id"] == "reasoning-123" - assert "summary" in reasoning_item, "summary field should be preserved" - assert "role" in reasoning_item, "role field should be preserved" - - # These fields should be filtered out if they are in the registry - all_excluded_fields = self.registry.get_all_excluded_fields() - if 'status' in all_excluded_fields: - assert "status" not in reasoning_item, "status field should be filtered out" - if 'content' in all_excluded_fields: - assert "content" not in reasoning_item, "content field should be filtered out" - if 'encrypted_content' in all_excluded_fields: - assert "encrypted_content" not in reasoning_item, "encrypted_content field should be filtered out" - - # Check the message item (index 2) - non-None values should be preserved - message_item = result[2] - assert message_item["type"] == "message" - assert message_item["status"] == "completed", "Non-None status should be preserved" - assert message_item["content"] == "Hello! How can I help?", "Non-None content should be preserved" - - def test_field_filtering_with_empty_registry(self): - """Test that filtering works gracefully when no models are registered""" - # Create a fresh registry for this test - from litellm.llms.openai.responses.transformation import OpenAIFieldExclusionRegistry - - # Save the current state - original_models = OpenAIFieldExclusionRegistry._MODELS_REQUIRING_EXCLUSION.copy() - - try: - # Clear the registry - OpenAIFieldExclusionRegistry._MODELS_REQUIRING_EXCLUSION.clear() - - # Test data - test_input = [{ - "id": "test-123", - "status": None, - "content": None, - "other_field": "should be preserved" - }] - - # Process the input - result = self.config._validate_input_param(test_input) - - # With empty registry, nothing should be filtered (all fields preserved) - assert len(result) == 1 - item = result[0] - assert "status" in item, "With empty registry, status should be preserved" - assert "content" in item, "With empty registry, content should be preserved" - assert item["other_field"] == "should be preserved" - - finally: - # Restore the original state - OpenAIFieldExclusionRegistry._MODELS_REQUIRING_EXCLUSION = original_models - - def test_pydantic_v1_v2_compatibility(self): - """Test that the registry works with both Pydantic v1 and v2""" - from pydantic import BaseModel - from typing import Optional - - class CompatibilityTestModel(BaseModel): - id: str - optional_field: Optional[str] = None - required_field: str = "default" - - # Register the model - self.registry.register_model(CompatibilityTestModel) - - # Get excluded fields - excluded_fields = self.registry.get_excluded_fields_for_model(CompatibilityTestModel) - - # Should work regardless of Pydantic version - assert isinstance(excluded_fields, set), "Should return a set" - assert 'optional_field' in excluded_fields, "Field with default None should be excluded" - - # Test that the model fields are accessible (works in both v1 and v2) - model_fields = getattr(CompatibilityTestModel, "model_fields", None) - if model_fields is None: - model_fields = getattr(CompatibilityTestModel, "__fields__", {}) - assert len(model_fields) > 0, "Should be able to access model fields" - - def test_non_registered_model_returns_empty_set(self): - """Test that non-registered models return empty excluded fields""" - from pydantic import BaseModel - - class UnregisteredModel(BaseModel): - id: str - some_field: str = None - - # Don't register this model - excluded_fields = self.registry.get_excluded_fields_for_model(UnregisteredModel) - - assert excluded_fields == set(), "Non-registered model should return empty set" - - @pytest.mark.parametrize("field_value", [None, "", 0, False, []]) - def test_only_none_values_are_filtered(self, field_value): - """Test that only None values are filtered, not other falsy values""" - test_input = [{ - "id": "test-123", - "status": field_value, - "content": "actual content", - "other_field": "preserved" - }] - - result = self.config._validate_input_param(test_input) - item = result[0] - - if field_value is None: - # Only None should be filtered (if status is in the registry) - all_excluded_fields = self.registry.get_all_excluded_fields() - if 'status' in all_excluded_fields: - assert "status" not in item, f"None value should be filtered out" - else: - assert item["status"] is None, f"If not in registry, None should be preserved" - else: - # Other falsy values should be preserved - assert "status" in item, f"Non-None value {field_value} should be preserved" - assert item["status"] == field_value, f"Value should be exactly {field_value}" + assert "stream" in params \ No newline at end of file From 231d54741f18588f5795beb8c3c3dc7c850df931 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 5 Sep 2025 10:40:26 +0530 Subject: [PATCH 5/5] Move test to test_litellm/ folder --- .../responses/test_text_format_conversion.py | 594 +----------------- 1 file changed, 11 insertions(+), 583 deletions(-) diff --git a/tests/test_litellm/responses/test_text_format_conversion.py b/tests/test_litellm/responses/test_text_format_conversion.py index dafccfaf2b9..20a87a4abbb 100644 --- a/tests/test_litellm/responses/test_text_format_conversion.py +++ b/tests/test_litellm/responses/test_text_format_conversion.py @@ -3,16 +3,11 @@ import os import sys import pytest +from pydantic import BaseModel sys.path.insert( - 0, os.path.abspath("../..") + 0, os.path.abspath("../../..") ) # Adds the parent directory to the system path -import json -from abc import ABC, abstractmethod - -from openai.types.responses.response_create_params import ( - ResponseInputParam, -) import litellm from litellm.types.llms.openai import ( @@ -22,512 +17,16 @@ from litellm.types.llms.openai import ( ) -def validate_responses_api_response(response, final_chunk: bool = False): - """ - Validate that a response from litellm.responses() or litellm.aresponses() - conforms to the expected ResponsesAPIResponse structure. +class TestTextFormatConversion: + """Test text_format to text parameter conversion for responses API""" - Args: - response: The response object to validate - - Raises: - AssertionError: If the response doesn't match the expected structure - """ - # Validate response structure - print("response=", json.dumps(response, indent=4, default=str)) - assert isinstance( - response, ResponsesAPIResponse - ), "Response should be an instance of ResponsesAPIResponse" - - # Required fields - assert "id" in response and isinstance( - response["id"], str - ), "Response should have a string 'id' field" - assert "created_at" in response and isinstance( - response["created_at"], int - ), "Response should have an integer 'created_at' field" - assert "output" in response and isinstance( - response["output"], list - ), "Response should have a list 'output' field" - assert "parallel_tool_calls" in response and isinstance( - response["parallel_tool_calls"], bool - ), "Response should have a boolean 'parallel_tool_calls' field" - - # Optional fields with their expected types - optional_fields = { - "error": (dict, type(None)), # error can be dict or None - "incomplete_details": (IncompleteDetails, type(None)), - "instructions": (str, type(None)), - "metadata": dict, - "model": str, - "object": str, - "temperature": (int, float, type(None)), - "tool_choice": (dict, str), - "tools": list, - "top_p": (int, float, type(None)), - "max_output_tokens": (int, type(None)), - "previous_response_id": (str, type(None)), - "reasoning": dict, - "status": str, - "text": dict, - "truncation": (str, type(None)), - "usage": ResponseAPIUsage, - "user": (str, type(None)), - "store": (bool, type(None)), - } - if final_chunk is False: - optional_fields["usage"] = type(None) - - for field, expected_type in optional_fields.items(): - if field in response: - assert isinstance( - response[field], expected_type - ), f"Field '{field}' should be of type {expected_type}, but got {type(response[field])}" - - # Check if output has at least one item - if final_chunk is True: - assert ( - len(response["output"]) > 0 - ), "Response 'output' field should have at least one item" - - return True # Return True if validation passes - - -class BaseResponsesAPITest(ABC): - """ - Abstract base test class that enforces a common test across all test classes. - """ - - @abstractmethod - def get_base_completion_call_args(self) -> dict: - """Must return the base completion call args""" - pass - - def get_base_completion_reasoning_call_args(self) -> dict: - """Must return the base completion reasoning call args""" - return None - - @pytest.mark.parametrize("sync_mode", [True, False]) - @pytest.mark.asyncio - async def test_basic_openai_responses_api(self, sync_mode): - litellm._turn_on_debug() - litellm.set_verbose = True - base_completion_call_args = self.get_base_completion_call_args() - try: - if sync_mode: - response = litellm.responses( - input="Basic ping", - max_output_tokens=20, - **base_completion_call_args, - ) - else: - response = await litellm.aresponses( - input="Basic ping", - max_output_tokens=20, - **base_completion_call_args, - ) - except litellm.InternalServerError: - pytest.skip("Skipping test due to litellm.InternalServerError") - print("litellm response=", json.dumps(response, indent=4, default=str)) - - # Use the helper function to validate the response - validate_responses_api_response(response, final_chunk=True) - - @pytest.mark.parametrize("sync_mode", [True, False]) - @pytest.mark.asyncio - @pytest.mark.flaky(retries=3, delay=2) - async def test_basic_openai_responses_api_streaming(self, sync_mode): - litellm._turn_on_debug() - base_completion_call_args = self.get_base_completion_call_args() - collected_content_string = "" - response_completed_event = None - if sync_mode: - response = litellm.responses( - input="Basic ping", stream=True, **base_completion_call_args - ) - for event in response: - print("litellm response=", json.dumps(event, indent=4, default=str)) - if event.type == "response.output_text.delta": - collected_content_string += event.delta - elif event.type == "response.completed": - response_completed_event = event - else: - response = await litellm.aresponses( - input="Basic ping", stream=True, **base_completion_call_args - ) - async for event in response: - print("litellm response=", json.dumps(event, indent=4, default=str)) - if event.type == "response.output_text.delta": - collected_content_string += event.delta - elif event.type == "response.completed": - response_completed_event = event - - # assert the delta chunks content had len(collected_content_string) > 0 - # this content is typically rendered on chat ui's - assert len(collected_content_string) > 0 - - # assert the response completed event is not None - assert response_completed_event is not None - - # assert the response completed event has a response - assert response_completed_event.response is not None - - # assert the response completed event includes the usage - assert response_completed_event.response.usage is not None - - # basic test assert the usage seems reasonable - print( - "response_completed_event.response.usage=", - response_completed_event.response.usage, - ) - assert ( - response_completed_event.response.usage.input_tokens > 0 - and response_completed_event.response.usage.input_tokens < 100 - ) - assert ( - response_completed_event.response.usage.output_tokens > 0 - and response_completed_event.response.usage.output_tokens < 2000 - ) - assert ( - response_completed_event.response.usage.total_tokens > 0 - and response_completed_event.response.usage.total_tokens < 2000 - ) - - # total tokens should be the sum of input and output tokens - assert ( - response_completed_event.response.usage.total_tokens - == response_completed_event.response.usage.input_tokens - + response_completed_event.response.usage.output_tokens - ) - - @pytest.mark.parametrize("sync_mode", [False, True]) - @pytest.mark.asyncio - async def test_basic_openai_responses_delete_endpoint(self, sync_mode): - litellm._turn_on_debug() - litellm.set_verbose = True - base_completion_call_args = self.get_base_completion_call_args() - if sync_mode: - response = litellm.responses( - input="Basic ping", max_output_tokens=20, **base_completion_call_args - ) - - # delete the response - if isinstance(response, ResponsesAPIResponse): - litellm.delete_responses( - response_id=response.id, **base_completion_call_args - ) - else: - raise ValueError("response is not a ResponsesAPIResponse") - else: - response = await litellm.aresponses( - input="Basic ping", max_output_tokens=20, **base_completion_call_args - ) - - # async delete the response - if isinstance(response, ResponsesAPIResponse): - await litellm.adelete_responses( - response_id=response.id, **base_completion_call_args - ) - else: - raise ValueError("response is not a ResponsesAPIResponse") - - @pytest.mark.parametrize("sync_mode", [True, False]) - @pytest.mark.flaky(retries=3, delay=2) - @pytest.mark.asyncio - async def test_basic_openai_responses_streaming_delete_endpoint(self, sync_mode): - # litellm._turn_on_debug() - # litellm.set_verbose = True - base_completion_call_args = self.get_base_completion_call_args() - response_id = None - if sync_mode: - response_id = None - response = litellm.responses( - input="Basic ping", - max_output_tokens=20, - stream=True, - **base_completion_call_args, - ) - for event in response: - print("litellm response=", json.dumps(event, indent=4, default=str)) - if "response" in event: - response_obj = event.get("response") - if response_obj is not None: - response_id = response_obj.get("id") - print("got response_id=", response_id) - - # delete the response - assert response_id is not None - litellm.delete_responses( - response_id=response_id, **base_completion_call_args - ) - else: - response = await litellm.aresponses( - input="Basic ping", - max_output_tokens=20, - stream=True, - **base_completion_call_args, - ) - async for event in response: - print("litellm response=", json.dumps(event, indent=4, default=str)) - if "response" in event: - response_obj = event.get("response") - if response_obj is not None: - response_id = response_obj.get("id") - print("got response_id=", response_id) - - # delete the response - assert response_id is not None - await litellm.adelete_responses( - response_id=response_id, **base_completion_call_args - ) - - @pytest.mark.parametrize("sync_mode", [False, True]) - @pytest.mark.flaky(retries=3, delay=2) - @pytest.mark.asyncio - async def test_basic_openai_responses_get_endpoint(self, sync_mode): - litellm._turn_on_debug() - litellm.set_verbose = True - base_completion_call_args = self.get_base_completion_call_args() - if sync_mode: - response = litellm.responses( - input="Basic ping", max_output_tokens=20, **base_completion_call_args - ) - - # get the response - if isinstance(response, ResponsesAPIResponse): - result = litellm.get_responses( - response_id=response.id, **base_completion_call_args - ) - assert result is not None - assert result.id == response.id - assert result.output == response.output - else: - raise ValueError("response is not a ResponsesAPIResponse") - else: - response = await litellm.aresponses( - input="Basic ping", max_output_tokens=20, **base_completion_call_args - ) - # async get the response - if isinstance(response, ResponsesAPIResponse): - result = await litellm.aget_responses( - response_id=response.id, **base_completion_call_args - ) - assert result is not None - assert result.id == response.id - assert result.output == response.output - else: - raise ValueError("response is not a ResponsesAPIResponse") - - @pytest.mark.asyncio - @pytest.mark.flaky(retries=3, delay=2) - async def test_basic_openai_list_input_items_endpoint(self): - """Test that calls the OpenAI List Input Items endpoint""" - litellm._turn_on_debug() - - response = await litellm.aresponses( - model="gpt-4o", - input="Tell me a three sentence bedtime story about a unicorn.", - ) - print("Initial response=", json.dumps(response, indent=4, default=str)) - - response_id = response.get("id") - assert response_id is not None, "Response should have an ID" - print(f"Got response_id: {response_id}") - - list_items_response = await litellm.alist_input_items( - response_id=response_id, - limit=20, - order="desc", - ) - print( - "List items response=", - json.dumps(list_items_response, indent=4, default=str), - ) - - @pytest.mark.asyncio - async def test_multiturn_responses_api(self): - litellm._turn_on_debug() - litellm.set_verbose = True - try: - base_completion_call_args = self.get_base_completion_call_args() - response_1 = await litellm.aresponses( - input="Basic ping", max_output_tokens=20, **base_completion_call_args - ) - - # follow up with a second request - response_1_id = response_1.id - response_2 = await litellm.aresponses( - input="Basic ping", - max_output_tokens=20, - previous_response_id=response_1_id, - **base_completion_call_args, - ) - - # assert the response is not None - assert response_1 is not None - assert response_2 is not None - except litellm.InternalServerError: - pytest.skip("Skipping test due to litellm.InternalServerError") - - @pytest.mark.asyncio - async def test_responses_api_with_tool_calls(self): - """Test that calls the Responses API with tool calls including function call and output""" - litellm._turn_on_debug() - litellm.set_verbose = True - base_completion_call_args = self.get_base_completion_call_args() - - # Define the input with message, function call, and function call output - input_data: ResponseInputParam = [ - { - "type": "message", - "role": "user", - "content": "How is the weather in São Paulo today ?", - }, - { - "type": "function_call", - "arguments": '{"location": "São Paulo, Brazil"}', - "call_id": "fc_1fe70e2a-a596-45ef-b72c-9b8567c460e5", - "name": "get_weather", - "id": "fc_1fe70e2a-a596-45ef-b72c-9b8567c460e5", - "status": "completed", - }, - { - "type": "function_call_output", - "call_id": "fc_1fe70e2a-a596-45ef-b72c-9b8567c460e5", - "output": "Rainy", - }, - ] - - # Define the tools - tools = [ - { - "type": "function", - "name": "get_weather", - "description": "Get current temperature for a given location.", - "parameters": { - "type": "object", - "properties": { - "location": { - "type": "string", - "description": "City and country e.g. Bogotá, Colombia", - } - }, - "required": ["location"], - "additionalProperties": False, - }, - } - ] - - try: - # Make the responses API call - response = await litellm.aresponses( - input=input_data, store=False, tools=tools, **base_completion_call_args - ) - except litellm.InternalServerError: - pytest.skip("Skipping test due to litellm.InternalServerError") - - print("litellm response=", json.dumps(response, indent=4, default=str)) - - # Validate the response structure - validate_responses_api_response(response, final_chunk=True) - - # Additional assertions specific to tool calls - assert response is not None - assert "output" in response - assert len(response["output"]) > 0 - - @pytest.mark.asyncio - async def test_responses_api_multi_turn_with_reasoning_and_structured_output(self): - """ - Test multi-turn conversation with reasoning, structured output, and tool calls. - - This test validates: - - First call: Model uses reasoning to process a question and makes a tool call - - Tool call handling: Function call output is properly processed - - Second call: Model produces structured output incorporating tool results - - Structured output: Response conforms to defined Pydantic model schema - """ - from pydantic import BaseModel - - litellm._turn_on_debug() - litellm.set_verbose = True - base_completion_call_args = self.get_base_completion_reasoning_call_args() - if base_completion_call_args is None: - pytest.skip("Skipping test due to no base completion reasoning call args") - - # Define tools for the conversation - tools = [{"type": "function", "name": "get_today"}] - - # Define structured output schema - class Output(BaseModel): - today: str - number_of_r: str - - # Initial conversation input - input_messages = [ - { - "role": "user", - "content": "How many r in strrawberrry? While you're thinking, you should call tool get_today. Then you output the today and number of r", - } - ] - - # First call - should trigger reasoning and tool call - response = await litellm.aresponses( - input=input_messages, - tools=tools, - reasoning={"effort": "low", "summary": "detailed"}, - text_format=Output, - **base_completion_call_args, - ) - - print("First call output:") - print(json.dumps(response.output, indent=4, default=str)) - - # Validate first response structure - validate_responses_api_response(response, final_chunk=True) - assert response.output is not None - assert len(response.output) > 0 - - # Extend input with first response output - input_messages.extend(response.output) - - # Process any tool calls and add function outputs - function_outputs = [] - for item in response.output: - if hasattr(item, "type") and item.type in [ - "function_call", - "custom_tool_call", - ]: - if hasattr(item, "name") and item.name == "get_today": - function_outputs.append( - { - "type": "function_call_output", - "call_id": item.call_id, - "output": "2025-01-15", - } - ) - - # Add function outputs to conversation - input_messages.extend(function_outputs) - - print("Second call input:") - print(json.dumps(input_messages, indent=4, default=str)) - - # Second call - should produce structured output - final_response = await litellm.aresponses( - input=input_messages, - tools=tools, - reasoning={"effort": "low", "summary": "detailed"}, - text_format=Output, - **base_completion_call_args, - ) - - print("Second call output:") - print(json.dumps(final_response.output, indent=4, default=str)) - - # Validate final response structure - validate_responses_api_response(final_response, final_chunk=True) - assert final_response.output is not None + def get_base_completion_call_args(self): + """Get base arguments for completion call""" + return { + "model": "gpt-4o", + "api_key": "test-key", + "api_base": "https://api.openai.com/v1", + } @pytest.mark.asyncio async def test_text_format_to_text_conversion(self): @@ -537,8 +36,6 @@ class BaseResponsesAPITest(ABC): """ from unittest.mock import AsyncMock, patch - from pydantic import BaseModel - class TestResponse(BaseModel): """Test Pydantic model for structured output""" @@ -662,72 +159,3 @@ class BaseResponsesAPITest(ABC): # Validate the response print("Response:", json.dumps(response, indent=4, default=str)) - - def test_openai_responses_api_dict_input_filtering(self): - """ - Test that regular dict inputs with status fields are properly filtered - to replicate exclude_unset=True behavior for non-Pydantic objects. - """ - from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig - - # Test input with regular dict objects (like from JSON) - test_input = [ - { - "role": "user", - "content": "test" - }, - { - "id": "rs_123", - "summary": [{"text": "test", "type": "summary_text"}], - "type": "reasoning", - "content": None, # Should be filtered out - "encrypted_content": None, # Should be filtered out - "status": None # Should be filtered out - }, - { - "arguments": "{}", - "call_id": "call_123", - "name": "get_today", - "type": "function_call", - "id": "fc_123", - "status": "completed" # Should be preserved (not a default field) - } - ] - - config = OpenAIResponsesAPIConfig() - validated_input = config._validate_input_param(test_input) - - # Verify the results - assert len(validated_input) == 3 - - # Check reasoning item (index 1) - reasoning_item = validated_input[1] - assert reasoning_item["type"] == "reasoning" - assert "status" not in reasoning_item, "status field should be filtered out from reasoning item" - assert "content" not in reasoning_item, "content field should be filtered out from reasoning item" - assert "encrypted_content" not in reasoning_item, "encrypted_content field should be filtered out from reasoning item" - assert "id" in reasoning_item, "id field should be preserved" - assert "summary" in reasoning_item, "summary field should be preserved" - - # Check function call item (index 2) - function_call_item = validated_input[2] - assert function_call_item["type"] == "function_call" - assert "status" in function_call_item, "status field should be preserved in function call item" - assert function_call_item["status"] == "completed", "status value should be preserved" - - print("✅ OpenAI Responses API dict input filtering test passed") - - -class TestOpenAIResponsesAPITest(BaseResponsesAPITest): - """Concrete test class for OpenAI Responses API tests""" - - def get_base_completion_call_args(self): - return { - "model": "openai/gpt-4o", - } - - def get_base_completion_reasoning_call_args(self): - return { - "model": "openai/gpt-5-mini", - } -