diff --git a/tests/llm_responses_api_testing/test_anthropic_responses_api.py b/tests/llm_responses_api_testing/test_anthropic_responses_api.py index 68ff22e8938..4b04e07c178 100644 --- a/tests/llm_responses_api_testing/test_anthropic_responses_api.py +++ b/tests/llm_responses_api_testing/test_anthropic_responses_api.py @@ -1,31 +1,10 @@ import os import sys import pytest -import asyncio -from typing import Optional -from unittest.mock import patch, AsyncMock, MagicMock -from litellm.responses.litellm_completion_transformation.handler import ( - LiteLLMCompletionTransformationHandler, -) -from litellm.responses.litellm_completion_transformation.transformation import ( - LiteLLMCompletionResponsesConfig, -) -from litellm.types.utils import ModelResponse sys.path.insert(0, os.path.abspath("../..")) import litellm -from litellm.integrations.custom_logger import CustomLogger -import json -from litellm.types.utils import StandardLoggingPayload -from litellm.types.llms.openai import ( - ResponseCompletedEvent, - ResponsesAPIResponse, - ResponseAPIUsage, - IncompleteDetails, -) -import litellm -from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler from base_responses_api import BaseResponsesAPITest from openai.types.responses.function_tool import FunctionTool @@ -130,87 +109,3 @@ def test_multiturn_tool_calls(): print("follow_up_response=", follow_up_response) -def test_response_api_handler_merges_metadata_and_service_tier_without_error(): - """Sync path must merge kwargs like async; double-splat raises TypeError.""" - handler = LiteLLMCompletionTransformationHandler() - - with patch("litellm.completion", new_callable=MagicMock) as mock_completion: - mock_completion.return_value = ModelResponse( - id="id", created=0, model="test", object="chat.completion", choices=[] - ) - handler.response_api_handler( - model="test", - input="hi", - responses_api_request={}, - metadata={"trace": "abc"}, - service_tier="auto", - ) - assert mock_completion.call_count == 1 - assert mock_completion.call_args.kwargs["metadata"] == {"trace": "abc"} - assert mock_completion.call_args.kwargs["service_tier"] == "auto" - - -@pytest.mark.asyncio -async def test_async_response_api_handler_merges_trace_id_without_error(): - handler = LiteLLMCompletionTransformationHandler() - - async def fake_session_handler(previous_response_id, litellm_completion_request): - litellm_completion_request["litellm_trace_id"] = "session-trace" - return litellm_completion_request - - with patch.object( - LiteLLMCompletionResponsesConfig, - "async_responses_api_session_handler", - side_effect=fake_session_handler, - ): - with patch("litellm.acompletion", new_callable=AsyncMock) as mock_acompletion: - mock_acompletion.return_value = ModelResponse( - id="id", created=0, model="test", object="chat.completion", choices=[] - ) - await handler.async_response_api_handler( - litellm_completion_request={"model": "test"}, - request_input="hi", - responses_api_request={"previous_response_id": "123"}, - litellm_trace_id="original-trace", - ) - # ensure acompletion called once with merged trace_id - assert mock_acompletion.call_count == 1 - assert ( - mock_acompletion.call_args.kwargs["litellm_trace_id"] == "session-trace" - ) - - -@pytest.mark.asyncio -async def test_aresponses_forwards_timeout_to_acompletion(): - """Regression test: timeout passed to aresponses() must reach acompletion() - on the completion transformation path (Anthropic, Bedrock, Vertex etc.). - - Previously, `timeout` was a named param of `responses()` but was NOT - forwarded to `litellm_completion_transformation_handler.response_api_handler`, - so it was silently dropped — `Router(timeout=N)` was a no-op for Anthropic - and similar providers, with calls falling back to the provider SDK default - (~600s for Anthropic). - """ - with patch("litellm.acompletion", new_callable=AsyncMock) as mock_acompletion: - mock_acompletion.return_value = ModelResponse( - id="id", - created=0, - model="anthropic/claude-sonnet-4-5", - object="chat.completion", - choices=[], - ) - - await litellm.aresponses( - model="anthropic/claude-sonnet-4-5", - input="hello", - timeout=42, - api_key="sk-ant-fake", - ) - - assert mock_acompletion.call_count == 1 - forwarded_timeout = mock_acompletion.call_args.kwargs.get("timeout") - assert forwarded_timeout == 42, ( - f"timeout was not forwarded to acompletion (got {forwarded_timeout!r}); " - "this means Router(timeout=N) silently fails for providers on the " - "completion transformation path." - ) diff --git a/tests/llm_responses_api_testing/test_openai_responses_api.py b/tests/llm_responses_api_testing/test_openai_responses_api.py index ea8b8fa886c..f4a940722be 100644 --- a/tests/llm_responses_api_testing/test_openai_responses_api.py +++ b/tests/llm_responses_api_testing/test_openai_responses_api.py @@ -13,15 +13,12 @@ import json sys.path.insert(0, os.path.abspath("../..")) import litellm from litellm.integrations.custom_logger import CustomLogger -import json from litellm.types.utils import StandardLoggingPayload from litellm.types.llms.openai import ( ResponseCompletedEvent, ResponsesAPIResponse, ResponseAPIUsage, - IncompleteDetails, ) -from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler from base_responses_api import BaseResponsesAPITest, validate_responses_api_response @@ -688,180 +685,6 @@ async def test_openai_responses_litellm_router_no_metadata(): mock_post.assert_called_once() -@pytest.mark.asyncio -async def test_openai_responses_litellm_router_with_metadata(): - """ - Test that metadata is correctly passed through when explicitly provided to the Router for responses API - """ - test_metadata = { - "user_id": "123", - "conversation_id": "abc", - "custom_field": "test_value", - } - - mock_response = { - "id": "resp_123", - "object": "response", - "created_at": 1741476542, - "status": "completed", - "model": "gpt-5.5", - "output": [ - { - "type": "message", - "id": "msg_123", - "status": "completed", - "role": "assistant", - "content": [ - {"type": "output_text", "text": "Hello world!", "annotations": []} - ], - } - ], - "parallel_tool_calls": True, - "usage": { - "input_tokens": 10, - "output_tokens": 20, - "total_tokens": 30, - "output_tokens_details": {"reasoning_tokens": 0}, - }, - "text": {"format": {"type": "text"}}, - "error": None, - "incomplete_details": None, - "instructions": None, - "metadata": test_metadata, # Include the test metadata in response - "temperature": 1.0, - "tool_choice": "auto", - "tools": [], - "top_p": 1.0, - "max_output_tokens": None, - "previous_response_id": None, - "reasoning": {"effort": None, "summary": None}, - "truncation": "disabled", - "user": None, - } - - class MockResponse: - def __init__(self, json_data, status_code): - self._json_data = json_data - self.status_code = status_code - self.text = str(json_data) - self.headers = httpx.Headers({}) - - def json(self): - return self._json_data - - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - new_callable=AsyncMock, - ) as mock_post: - # Configure the mock to return our response - mock_post.return_value = MockResponse(mock_response, 200) - - litellm._turn_on_debug() - router = litellm.Router( - model_list=[ - { - "model_name": "gpt4o-special-alias", - "litellm_params": { - "model": "gpt-5.5", - "api_key": "fake-key", - }, - } - ] - ) - - # Call the handler with metadata - await router.aresponses( - model="gpt4o-special-alias", - input="Hello, can you tell me a short joke?", - metadata=test_metadata, - ) - - # Check the request body - request_body = mock_post.call_args.kwargs["json"] - print("Request body:", json.dumps(request_body, indent=4)) - - # Assert metadata matches exactly what was passed - assert ( - request_body["metadata"] == test_metadata - ), "metadata in request body should match what was passed" - mock_post.assert_called_once() - - -@pytest.mark.asyncio -async def test_openai_responses_litellm_router_with_prompt(): - """Test that prompt object is passed through the Router for responses API""" - - prompt_obj = { - "id": "pmpt_abc123", - "version": "2", - "variables": {"random_variable": "ishaan_from_litellm"}, - } - - mock_response = { - "id": "resp_123", - "object": "response", - "created_at": 1741476542, - "status": "completed", - "model": "gpt-5.5", - "output": [], - "parallel_tool_calls": True, - "usage": {"input_tokens": 0, "output_tokens": 0, "total_tokens": 0}, - "text": {"format": {"type": "text"}}, - "error": None, - "incomplete_details": None, - "instructions": None, - "metadata": {}, - "temperature": 1.0, - "tool_choice": "auto", - "tools": [], - "top_p": 1.0, - "max_output_tokens": None, - "previous_response_id": None, - "reasoning": {"effort": None, "summary": None}, - "truncation": "disabled", - "user": None, - } - - class MockResponse: - def __init__(self, json_data, status_code): - self._json_data = json_data - self.status_code = status_code - self.text = str(json_data) - self.headers = httpx.Headers({}) - - def json(self): - return self._json_data - - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - new_callable=AsyncMock, - ) as mock_post: - mock_post.return_value = MockResponse(mock_response, 200) - - litellm._turn_on_debug() - router = litellm.Router( - model_list=[ - { - "model_name": "gpt4o-special-alias", - "litellm_params": { - "model": "gpt-5.5", - "api_key": "fake-key", - }, - } - ] - ) - - await router.aresponses( - model="gpt4o-special-alias", - input="Hello", - prompt=prompt_obj, - ) - - request_body = mock_post.call_args.kwargs["json"] - assert request_body["prompt"] == prompt_obj - mock_post.assert_called_once() - - def test_bad_request_bad_param_error(): """Raise a BadRequestError when an invalid parameter value is provided""" try: @@ -1111,106 +934,6 @@ async def test_openai_o1_pro_response_api_streaming(sync_mode): assert "stream" not in request_body -def test_basic_computer_use_preview_tool_call(): - """ - Test that LiteLLM correctly handles a computer_use_preview tool call where the environment is set to "linux" - - linux is an unsupported environment for the computer_use_preview tool, but litellm users should still be able to pass it to openai - """ - # Mock response from OpenAI - - mock_response = { - "id": "resp_67dc3dd77b388190822443a85252da5a0e13d8bdc0e28d88", - "object": "response", - "created_at": 1742486999, - "status": "incomplete", - "error": None, - "incomplete_details": {"reason": "max_output_tokens"}, - "instructions": None, - "max_output_tokens": 20, - "model": "o1-pro-2025-03-19", - "output": [ - { - "type": "reasoning", - "id": "rs_67dc3de50f64819097450ed50a33d5f90e13d8bdc0e28d88", - "summary": [], - } - ], - "parallel_tool_calls": True, - "previous_response_id": None, - "reasoning": {"effort": "medium", "generate_summary": None}, - "store": True, - "temperature": 1.0, - "text": {"format": {"type": "text"}}, - "tool_choice": "auto", - "tools": [], - "top_p": 1.0, - "truncation": "disabled", - "usage": { - "input_tokens": 73, - "input_tokens_details": {"cached_tokens": 0}, - "output_tokens": 20, - "output_tokens_details": {"reasoning_tokens": 0}, - "total_tokens": 93, - }, - "user": None, - "metadata": {}, - } - - class MockResponse: - def __init__(self, json_data, status_code): - self._json_data = json_data - self.status_code = status_code - self.text = json.dumps(json_data) - self.headers = httpx.Headers({}) - - def json(self): - return self._json_data - - with patch( - "litellm.llms.custom_httpx.http_handler.HTTPHandler.post", - return_value=MockResponse(mock_response, 200), - ) as mock_post: - litellm._turn_on_debug() - litellm.set_verbose = True - - # Call the responses API with computer_use_preview tool - response = litellm.responses( - model="openai/computer-use-preview", - tools=[ - { - "type": "computer_use_preview", - "display_width": 1024, - "display_height": 768, - "environment": "linux", # other possible values: "mac", "windows", "ubuntu" - } - ], - input="Check the latest OpenAI news on bing.com.", - reasoning={"summary": "concise"}, - truncation="auto", - ) - - # Verify the request was made correctly - mock_post.assert_called_once() - request_body = mock_post.call_args.kwargs["json"] - - # Validate the request structure - assert request_body["model"] == "computer-use-preview" - assert len(request_body["tools"]) == 1 - assert request_body["tools"][0]["type"] == "computer_use_preview" - assert request_body["tools"][0]["display_width"] == 1024 - assert request_body["tools"][0]["display_height"] == 768 - assert request_body["tools"][0]["environment"] == "linux" - - # Check that reasoning was passed correctly - assert request_body["reasoning"]["summary"] == "concise" - assert request_body["truncation"] == "auto" - - # Validate the input format - assert isinstance(request_body["input"], str) - assert request_body["input"] == "Check the latest OpenAI news on bing.com." - - def test_mcp_tools_with_responses_api(): litellm._turn_on_debug() MCP_TOOLS = [ @@ -1418,193 +1141,6 @@ async def test_store_field_transformation(): ), "created_at should maintain the same value after conversion" -@pytest.mark.asyncio -async def test_aresponses_service_tier_and_safety_identifier(): - """ - Test that service_tier and safety_identifier parameters are correctly sent in the request body - when using litellm.aresponses. - """ - mock_response = { - "id": "resp_01234567890abcdef", - "object": "response", - "created_at": 1753060947, - "status": "completed", - "error": None, - "incomplete_details": None, - "instructions": None, - "max_output_tokens": None, - "model": "gpt-4o-2024-05-13", - "output": [ - { - "type": "text", - "id": "out_01234567890abcdef", - "text": "This is a test response with service tier and safety identifier.", - } - ], - "parallel_tool_calls": True, - "previous_response_id": None, - "reasoning": None, - "store": True, - "temperature": 1.0, - "text": {"format": {"type": "text"}}, - "tool_choice": "auto", - "tools": [], - "top_p": 1.0, - "truncation": "disabled", - "usage": { - "input_tokens": 15, - "input_tokens_details": {"cached_tokens": 0}, - "output_tokens": 25, - "output_tokens_details": {"reasoning_tokens": 0}, - "total_tokens": 40, - }, - "user": None, - "metadata": {}, - } - - class MockResponse: - def __init__(self, json_data, status_code): - self._json_data = json_data - self.status_code = status_code - self.text = json.dumps(json_data) - self.headers = httpx.Headers({}) - - def json(self): - return self._json_data - - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - new_callable=AsyncMock, - ) as mock_post: - # Configure the mock to return our response - mock_post.return_value = MockResponse(mock_response, 200) - - litellm._turn_on_debug() - litellm.set_verbose = True - - # Call aresponses with service_tier and safety_identifier - response = await litellm.aresponses( - model="openai/gpt-5.5", - input="Test with service tier and safety identifier", - service_tier="flex", - safety_identifier="123", - ) - - # Verify the request was made correctly - mock_post.assert_called_once() - request_body = mock_post.call_args.kwargs["json"] - print("request_body=", json.dumps(request_body, indent=4, default=str)) - - # Validate that both parameters are present in the request body - assert ( - request_body["service_tier"] == "flex" - ), "service_tier should be 'flex' in request body" - assert ( - request_body["safety_identifier"] == "123" - ), "safety_identifier should be '123' in request body" - assert request_body["model"] == "gpt-5.5" - assert request_body["input"] == "Test with service tier and safety identifier" - - # Validate the response - print("Response:", json.dumps(response, indent=4, default=str)) - - -@pytest.mark.asyncio -async def test_openai_gpt5_reasoning_effort_parameter(): - """Test that reasoning_effort parameter is properly sent in the HTTP request for GPT-5 models.""" - - # Mock response for GPT-5 responses API (correct format) - mock_response = { - "id": "resp_01ABC123", - "object": "response", - "created_at": 1729621667, - "status": "completed", - "model": "gpt-5-mini", - "output": [ - { - "type": "message", - "id": "msg_123", - "status": "completed", - "role": "assistant", - "content": [ - { - "type": "output_text", - "text": "The capital of France is Paris.", - "annotations": [], - } - ], - } - ], - "parallel_tool_calls": True, - "usage": { - "input_tokens": 15, - "input_tokens_details": {"cached_tokens": 0}, - "output_tokens": 8, - "output_tokens_details": {"reasoning_tokens": 0}, - "total_tokens": 23, - }, - "text": {"format": {"type": "text"}}, - "error": None, - "incomplete_details": None, - "instructions": None, - "metadata": {}, - "temperature": 1.0, - "tool_choice": "auto", - "tools": [], - "top_p": 1.0, - "max_output_tokens": None, - "previous_response_id": None, - "reasoning": {"effort": "low", "summary": None}, - "truncation": "disabled", - "user": None, - } - - class MockResponse: - def __init__(self, json_data, status_code): - self._json_data = json_data - self.status_code = status_code - self.text = json.dumps(json_data) - self.headers = httpx.Headers({}) - - def json(self): - return self._json_data - - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - new_callable=AsyncMock, - ) as mock_post: - # Configure the mock to return our response - mock_post.return_value = MockResponse(mock_response, 200) - - litellm._turn_on_debug() - litellm.set_verbose = True - - # Call aresponses with reasoning_effort parameter - response = await litellm.aresponses( - model="openai/gpt-5-mini", - input="What is the capital of France?", - reasoning={"effort": "minimal"}, - ) - - # Verify the request was made correctly - mock_post.assert_called_once() - request_body = mock_post.call_args.kwargs["json"] - print("request_body=", json.dumps(request_body, indent=4, default=str)) - print("reasoning=", request_body["reasoning"]) - # Validate that reasoning_effort is present in the request body - assert ( - "reasoning" in request_body - ), "reasoning should be present in request body" - assert ( - request_body["reasoning"]["effort"] == "minimal" - ), "reasoning_effort should be 'minimal' in request body" - assert request_body["model"] == "gpt-5-mini" - assert request_body["input"] == "What is the capital of France?" - - # Validate the response - print("Response:", json.dumps(response, indent=4, default=str)) - - @pytest.mark.asyncio @pytest.mark.parametrize("stream", [True, False]) async def test_basic_openai_responses_with_websearch(stream): @@ -1737,95 +1273,6 @@ def extra_body_mock_response_data(): } -@pytest.mark.asyncio -async def test_aresponses_extra_body_params_passed(extra_body_mock_response_data): - """Test that extra_body parameters are passed in async mode.""" - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - new_callable=AsyncMock, - ) as mock_post: - mock_post.return_value = MockResponse(extra_body_mock_response_data, 200) - - response = await litellm.aresponses( - model="gpt-5.5", - input="Test input", - max_output_tokens=20, - extra_body={ - "custom_param_1": "value1", - "custom_param_2": {"nested": "value2"}, - "experimental_feature": True, - }, - ) - - assert response is not None - assert response.id is not None - - request_body = mock_post.call_args.kwargs["json"] - - assert "custom_param_1" in request_body - assert request_body["custom_param_1"] == "value1" - assert "custom_param_2" in request_body - assert request_body["custom_param_2"]["nested"] == "value2" - assert "experimental_feature" in request_body - assert request_body["experimental_feature"] is True - assert request_body["model"] == "gpt-5.5" - assert request_body["input"] == "Test input" - - -def test_responses_extra_body_params_passed_sync(extra_body_mock_response_data): - """Test that extra_body parameters are passed in sync mode.""" - with patch( - "litellm.llms.custom_httpx.http_handler.HTTPHandler.post", - return_value=MockResponse(extra_body_mock_response_data, 200), - ) as mock_post: - response = litellm.responses( - model="gpt-5.5", - input="Sync test", - max_output_tokens=20, - extra_body={ - "sync_custom_param": "sync_value", - "another_param": 42, - }, - ) - - assert response is not None - assert response.id is not None - - request_body = mock_post.call_args.kwargs["json"] - - assert "sync_custom_param" in request_body - assert request_body["sync_custom_param"] == "sync_value" - assert "another_param" in request_body - assert request_body["another_param"] == 42 - assert request_body["model"] == "gpt-5.5" - - -@pytest.mark.asyncio -async def test_extra_body_merges_with_request_data(extra_body_mock_response_data): - """Test that extra_body is merged into the request data.""" - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - new_callable=AsyncMock, - ) as mock_post: - mock_post.return_value = MockResponse(extra_body_mock_response_data, 200) - - await litellm.aresponses( - model="gpt-5.5", - input="Test", - temperature=0.7, - max_output_tokens=20, - extra_body={ - "custom_field": "custom_value", - }, - ) - - request_body = mock_post.call_args.kwargs["json"] - - assert "temperature" in request_body - assert "custom_field" in request_body - assert request_body["custom_field"] == "custom_value" - - @pytest.mark.asyncio @pytest.mark.parametrize("sync_mode", [True, False]) async def test_openai_compact_responses_api(sync_mode):