import contextlib import copy import json import os import sys import httpx import pytest import respx from fastapi.testclient import TestClient sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path import urllib.parse from unittest.mock import MagicMock, patch import litellm from litellm import main as litellm_main async def _async_fake_bedrock_image_details(image_url): return "ZmFrZS1pbWFnZQ==", "image/png" @pytest.fixture(autouse=True) def clear_client_cache(): """ Clear the HTTP client cache before each test to ensure mocks are used. This prevents cached real clients from being reused across tests. """ cache = getattr(litellm, "in_memory_llm_clients_cache", None) if cache is not None: cache.flush_cache() yield if cache is not None: cache.flush_cache() @pytest.fixture(autouse=True) def add_api_keys_to_env(monkeypatch): monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-api03-1234567890") monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-api03-1234567890") monkeypatch.setenv("AWS_ACCESS_KEY_ID", "my-fake-aws-access-key-id") monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "my-fake-aws-secret-access-key") monkeypatch.setenv("AWS_REGION", "us-east-1") # Keep these transformation tests on the simple access-key path. A leaked # session token or role/web-identity env var pushes Bedrock auth down a # different branch and fails before the mocked HTTP client is exercised. monkeypatch.delenv("AWS_SESSION_TOKEN", raising=False) monkeypatch.delenv("AWS_ROLE_ARN", raising=False) monkeypatch.delenv("AWS_WEB_IDENTITY_TOKEN_FILE", raising=False) @pytest.fixture def openai_api_response(): mock_response_data = { "id": "chatcmpl-B0W3vmiM78Xkgx7kI7dr7PC949DMS", "choices": [ { "finish_reason": "stop", "index": 0, "logprobs": None, "message": { "content": "", "refusal": None, "role": "assistant", "audio": None, "function_call": None, "tool_calls": None, }, } ], "created": 1739462947, "model": "gpt-4o-mini-2024-07-18", "object": "chat.completion", "service_tier": "default", "system_fingerprint": "fp_bd83329f63", "usage": { "completion_tokens": 1, "prompt_tokens": 121, "total_tokens": 122, "completion_tokens_details": { "accepted_prediction_tokens": 0, "audio_tokens": 0, "reasoning_tokens": 0, "rejected_prediction_tokens": 0, }, "prompt_tokens_details": {"audio_tokens": 0, "cached_tokens": 0}, }, } return mock_response_data def test_completion_missing_role(openai_api_response): from openai import OpenAI from litellm.types.utils import ModelResponse client = OpenAI(api_key="test_api_key") mock_raw_response = MagicMock() mock_raw_response.headers = { "x-request-id": "123", "openai-organization": "org-123", "x-ratelimit-limit-requests": "100", "x-ratelimit-remaining-requests": "99", } mock_raw_response.parse.return_value = ModelResponse(**openai_api_response) print(f"openai_api_response: {openai_api_response}") with patch.object( client.chat.completions.with_raw_response, "create", mock_raw_response ) as mock_create: litellm.completion( model="gpt-4o-mini", messages=[ {"role": "user", "content": "Hey"}, { "content": "", "tool_calls": [ { "id": "call_m0vFJjQmTH1McvaHBPR2YFwY", "function": { "arguments": '{"input": "dksjsdkjdhskdjshdskhjkhlk"}', "name": "tool_name", }, "type": "function", "index": 0, }, { "id": "call_Vw6RaqV2n5aaANXEdp5pYxo2", "function": { "arguments": '{"input": "jkljlkjlkjlkjlk"}', "name": "tool_name", }, "type": "function", "index": 1, }, { "id": "call_hBIKwldUEGlNh6NlSXil62K4", "function": { "arguments": '{"input": "jkjlkjlkjlkj;lj"}', "name": "tool_name", }, "type": "function", "index": 2, }, ], }, ], client=client, ) mock_create.assert_called_once() @pytest.mark.parametrize( "model", [ "gemini/gemini-1.5-flash", "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", "bedrock/invoke/anthropic.claude-haiku-4-5-20251001-v1:0", "anthropic/claude-3-5-sonnet", ], ) @pytest.mark.parametrize("sync_mode", [True, False]) @pytest.mark.asyncio async def test_url_with_format_param(model, sync_mode, monkeypatch): from litellm import acompletion, completion from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.litellm_core_utils.prompt_templates import factory as prompt_factory if sync_mode: client = HTTPHandler() else: client = AsyncHTTPHandler() # This test is about request shaping, not live image downloads. Stub the # URL->image conversion helpers so suite-level network/client state from # earlier tests cannot prevent the mocked provider client from being hit. fake_base64_image = "data:image/png;base64,ZmFrZS1pbWFnZQ==" monkeypatch.setattr( prompt_factory, "convert_url_to_base64", lambda url: fake_base64_image ) monkeypatch.setattr( prompt_factory.BedrockImageProcessor, "get_image_details", staticmethod(lambda image_url: ("ZmFrZS1pbWFnZQ==", "image/png")), ) monkeypatch.setattr( prompt_factory.BedrockImageProcessor, "get_image_details_async", staticmethod(_async_fake_bedrock_image_details), ) args = { "model": model, "messages": [ { "role": "user", "content": [ { "type": "image_url", "image_url": { "url": "https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/c233c9ade2ccb5491072ae232c814942.png", "format": "image/png", }, }, {"type": "text", "text": "Describe this image"}, ], } ], } if model.startswith("gemini/"): args["api_key"] = "test-api-key" with patch.object(client, "post", new=MagicMock()) as mock_client: try: if sync_mode: response = completion(**args, client=client) else: response = await acompletion(**args, client=client) print(response) except Exception as e: pass mock_client.assert_called() print(mock_client.call_args.kwargs) if "data" in mock_client.call_args.kwargs: json_str = mock_client.call_args.kwargs["data"] else: json_str = json.dumps(mock_client.call_args.kwargs["json"]) if isinstance(json_str, bytes): json_str = json_str.decode("utf-8") print(f"type of json_str: {type(json_str)}") # Bedrock models convert URLs to base64, while direct Anthropic models support URLs # bedrock/invoke models use Anthropic messages API which supports URLs if model.startswith("bedrock/invoke/"): # bedrock/invoke should convert URLs to base64 (doesn't support URL references) # URL should NOT be in the JSON (it should be converted to base64) assert "https://awsmp-logos.s3.amazonaws.com" not in json_str # Should have base64 data in the source (type="base64", not type="url") assert '"type":"base64"' in json_str or '"type": "base64"' in json_str # Should have "data" field containing base64 content assert '"data"' in json_str elif model.startswith("bedrock/"): # Regular Bedrock models should convert URLs to base64 (uses "bytes" field) # URL should NOT be in the JSON (it should be converted to base64) assert "https://awsmp-logos.s3.amazonaws.com" not in json_str # Should have "bytes" field (Bedrock uses "bytes" not "base64" in the field name) assert '"bytes"' in json_str or '"bytes":' in json_str elif model.startswith("anthropic/"): # Direct Anthropic models should pass HTTPS URLs directly (HTTP URLs are converted to base64) # Since we're using HTTPS URL, it should be passed as-is assert "https://awsmp-logos.s3.amazonaws.com" in json_str # For Anthropic, URL references use "url" type, not base64 assert '"type":"url"' in json_str or '"type": "url"' in json_str else: # For other models, check format parameter is respected assert "png" in json_str assert "jpeg" not in json_str @pytest.mark.parametrize("model", ["gpt-4o-mini"]) @pytest.mark.parametrize("sync_mode", [True, False]) @pytest.mark.asyncio async def test_url_with_format_param_openai(model, sync_mode): from openai import AsyncOpenAI, OpenAI from litellm import acompletion, completion if sync_mode: client = OpenAI() else: client = AsyncOpenAI() args = { "model": model, "messages": [ { "role": "user", "content": [ { "type": "image_url", "image_url": { "url": "https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/c233c9ade2ccb5491072ae232c814942.png", "format": "image/png", }, }, {"type": "text", "text": "Describe this image"}, ], } ], } with patch.object( client.chat.completions.with_raw_response, "create" ) as mock_client: try: if sync_mode: response = completion(**args, client=client) else: response = await acompletion(**args, client=client) print(response) except Exception as e: print(e) mock_client.assert_called() print(mock_client.call_args.kwargs) json_str = json.dumps(mock_client.call_args.kwargs) assert "format" not in json_str def test_bedrock_latency_optimized_inference(): from litellm.llms.custom_httpx.http_handler import HTTPHandler client = HTTPHandler() with patch.object(client, "post") as mock_post: try: response = litellm.completion( model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", messages=[{"role": "user", "content": "Hello, how are you?"}], performanceConfig={"latency": "optimized"}, client=client, ) except Exception as e: print(e) mock_post.assert_called_once() json_data = json.loads(mock_post.call_args.kwargs["data"]) assert json_data["performanceConfig"]["latency"] == "optimized" def test_strip_input_examples_for_non_anthropic_providers(): tools = [ { "type": "function", "name": "example_tool", "input_examples": [{"foo": "bar"}], "function": { "name": "example_tool", "input_examples": [{"foo": "bar"}], }, } ] assert not litellm_main._should_allow_input_examples( custom_llm_provider="openai", model="gpt-4o-mini" ) cleaned = litellm_main._drop_input_examples_from_tools(tools=tools) assert isinstance(cleaned, list) assert "input_examples" not in cleaned[0] assert "input_examples" not in cleaned[0]["function"] def test_custom_provider_with_extra_headers(): from litellm.llms.custom_httpx.http_handler import HTTPHandler with patch.object( litellm.llms.custom_httpx.http_handler.HTTPHandler, "post" ) as mock_post: response = litellm.completion( model="custom/custom", messages=[{"role": "user", "content": "Hello, how are you?"}], headers={"X-Custom-Header": "custom-value"}, api_base="https://example.com/api/v1", ) mock_post.assert_called_once() assert mock_post.call_args[1]["headers"]["X-Custom-Header"] == "custom-value" def test_custom_provider_with_extra_body(): from litellm.llms.custom_httpx.http_handler import HTTPHandler with patch.object( litellm.llms.custom_httpx.http_handler.HTTPHandler, "post" ) as mock_post: response = litellm.completion( model="custom/custom", messages=[{"role": "user", "content": "Hello, how are you?"}], extra_body={ "X-Custom-BodyValue": "custom-value", "X-Custom-BodyValue2": "custom-value2", }, api_base="https://example.com/api/v1", ) mock_post.assert_called_once() assert mock_post.call_args[1]["json"]["X-Custom-BodyValue"] == "custom-value" assert mock_post.call_args[1]["json"] == { "model": "custom", "params": { "prompt": ["Hello, how are you?"], "max_tokens": None, "temperature": None, "top_p": None, "top_k": None, }, "X-Custom-BodyValue": "custom-value", "X-Custom-BodyValue2": "custom-value2", } # test that extra_body is not passed if not provided with patch.object( litellm.llms.custom_httpx.http_handler.HTTPHandler, "post" ) as mock_post: response = litellm.completion( model="custom/custom", messages=[{"role": "user", "content": "Hello, how are you?"}], api_base="https://example.com/api/v1", ) mock_post.assert_called_once() assert mock_post.call_args[1]["json"] == { "model": "custom", "params": { "prompt": ["Hello, how are you?"], "max_tokens": None, "temperature": None, "top_p": None, "top_k": None, }, } @pytest.fixture(autouse=True) def set_openrouter_api_key(): original_api_key = os.environ.get("OPENROUTER_API_KEY") os.environ["OPENROUTER_API_KEY"] = "fake-key-for-testing" yield if original_api_key is not None: os.environ["OPENROUTER_API_KEY"] = original_api_key else: del os.environ["OPENROUTER_API_KEY"] @pytest.mark.asyncio async def test_extra_body_with_fallback( respx_mock: respx.MockRouter, set_openrouter_api_key, monkeypatch ): """ test regression for https://github.com/BerriAI/litellm/issues/8425. This was perhaps a wider issue with the acompletion function not passing kwargs such as extra_body correctly when fallbacks are specified. """ # Save original state to restore after test original_disable_aiohttp = litellm.disable_aiohttp_transport try: # since this uses respx, we need to set use_aiohttp_transport to False # Set both the global variable and environment variable to ensure it takes effect litellm.disable_aiohttp_transport = True monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") # Flush cache to ensure no stale aiohttp clients are used litellm.in_memory_llm_clients_cache.flush_cache() # Set up test parameters model = "openrouter/deepseek/deepseek-chat" messages = [{"role": "user", "content": "Hello, world!"}] extra_body = { "provider": { "order": ["DeepSeek"], "allow_fallbacks": False, "require_parameters": True, } } fallbacks = [{"model": "openrouter/google/gemini-flash-1.5-8b"}] # Set up mock to respond to any POST request to the OpenRouter endpoint # This ensures it works for both primary and fallback models mock_route = respx_mock.post("https://openrouter.ai/api/v1/chat/completions") mock_route.return_value = httpx.Response( 200, json={ "id": "chatcmpl-123", "object": "chat.completion", "created": 1677652288, "model": model, "choices": [ { "index": 0, "message": { "role": "assistant", "content": "Hello from mocked response!", }, "finish_reason": "stop", } ], "usage": { "prompt_tokens": 9, "completion_tokens": 12, "total_tokens": 21, }, }, ) response = await litellm.acompletion( model=model, messages=messages, extra_body=extra_body, fallbacks=fallbacks, api_key="fake-openrouter-api-key", ) # Verify the response assert response is not None assert ( len(respx_mock.calls) > 0 ), "Mock was not called - check if aiohttp transport is properly disabled" # Get the request from the mock request: httpx.Request = respx_mock.calls[0].request request_body = request.read() request_body = json.loads(request_body) # Verify basic parameters assert request_body["model"] == "deepseek/deepseek-chat" assert request_body["messages"] == messages # Verify the extra_body parameters remain under the provider key assert request_body["provider"]["order"] == ["DeepSeek"] assert request_body["provider"]["allow_fallbacks"] is False assert request_body["provider"]["require_parameters"] is True finally: # Restore original state to prevent test pollution litellm.disable_aiohttp_transport = original_disable_aiohttp litellm.in_memory_llm_clients_cache.flush_cache() @pytest.mark.parametrize("env_base", ["OPENAI_BASE_URL", "OPENAI_API_BASE"]) @pytest.mark.asyncio @pytest.mark.flaky(retries=3, delay=1) async def test_openai_env_base( respx_mock: respx.MockRouter, env_base, openai_api_response, monkeypatch ): "This tests OpenAI env variables are honored, including legacy OPENAI_API_BASE" # Ensure aiohttp transport is disabled to use httpx which respx can mock litellm.disable_aiohttp_transport = True expected_base_url = "http://localhost:12345/v1" # Assign the environment variable based on env_base, and use a fake API key. monkeypatch.setenv(env_base, expected_base_url) monkeypatch.setenv("OPENAI_API_KEY", "fake_openai_api_key") model = "gpt-4o" messages = [{"role": "user", "content": "Hello, how are you?"}] # Configure respx mock to intercept the request mock_route = respx_mock.post( url__regex=r"http://localhost:12345/v1/chat/completions.*" ).mock( return_value=httpx.Response( status_code=200, json={ "id": "chatcmpl-123", "object": "chat.completion", "created": 1677652288, "model": model, "choices": [ { "index": 0, "message": { "role": "assistant", "content": "Hello from mocked response!", }, "finish_reason": "stop", } ], "usage": { "prompt_tokens": 9, "completion_tokens": 12, "total_tokens": 21, }, }, ) ) try: response = await litellm.acompletion(model=model, messages=messages) # verify we had a response assert response.choices[0].message.content == "Hello from mocked response!" # Verify the mock was called assert ( mock_route.called ), "Mock route was not called - request may have bypassed respx" finally: # Clean up to avoid affecting other tests litellm.disable_aiohttp_transport = False def build_database_url(username, password, host, dbname): username_enc = urllib.parse.quote_plus(username) password_enc = urllib.parse.quote_plus(password) dbname_enc = urllib.parse.quote_plus(dbname) return f"postgresql://{username_enc}:{password_enc}@{host}/{dbname_enc}" def test_build_database_url(): url = build_database_url("user@name", "p@ss:word", "localhost", "db/name") assert url == "postgresql://user%40name:p%40ss%3Aword@localhost/db%2Fname" def test_bedrock_llama(): litellm._turn_on_debug() from litellm.types.utils import CallTypes from litellm.utils import return_raw_request model = "bedrock/invoke/us.meta.llama4-scout-17b-instruct-v1:0" request = return_raw_request( endpoint=CallTypes.completion, kwargs={ "model": model, "messages": [ {"role": "user", "content": "hi"}, ], }, ) print(request) assert ( request["raw_request_body"]["prompt"] == "<|begin_of_text|><|start_header_id|>user<|end_header_id|>\n\nhi<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n" ) def _mocked_openai_chat_response(model: str) -> httpx.Response: return httpx.Response( status_code=200, json={ "id": "chatcmpl-123", "object": "chat.completion", "created": 1677652288, "model": model, "choices": [ { "index": 0, "message": { "role": "assistant", "content": "Hello from mocked response!", }, "finish_reason": "stop", } ], "usage": { "prompt_tokens": 9, "completion_tokens": 12, "total_tokens": 21, }, }, ) def test_completion_forwards_verbosity_in_raw_request(respx_mock: respx.MockRouter): """Regression test: completion() must forward the verbosity param to the provider request body.""" from litellm.types.utils import CallTypes from litellm.utils import return_raw_request model = "gpt-5.2" messages = [{"role": "user", "content": "hi"}] respx_mock.post("https://api.openai.com/v1/chat/completions").mock( return_value=_mocked_openai_chat_response(model) ) request = return_raw_request( endpoint=CallTypes.completion, kwargs={ "model": model, "messages": messages, "verbosity": "high", }, ) assert request["raw_request_body"]["verbosity"] == "high" assert request["raw_request_body"]["model"] == model assert request["raw_request_body"]["messages"] == messages @pytest.mark.asyncio async def test_acompletion_forwards_verbosity_to_provider_request( respx_mock: respx.MockRouter, monkeypatch ): """Regression test: acompletion() must forward the verbosity param to the provider request body.""" original_disable_aiohttp = litellm.disable_aiohttp_transport try: litellm.disable_aiohttp_transport = True monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") litellm.in_memory_llm_clients_cache.flush_cache() model = "gpt-5.2" messages = [{"role": "user", "content": "hi"}] mock_route = respx_mock.post("https://api.openai.com/v1/chat/completions").mock( return_value=_mocked_openai_chat_response(model) ) response = await litellm.acompletion( model=model, messages=messages, verbosity="low", api_key="fake-openai-api-key", ) assert response.choices[0].message.content == "Hello from mocked response!" assert mock_route.called request_body = json.loads(respx_mock.calls[0].request.read()) assert request_body["verbosity"] == "low" assert request_body["model"] == model assert request_body["messages"] == messages finally: litellm.disable_aiohttp_transport = original_disable_aiohttp litellm.in_memory_llm_clients_cache.flush_cache() def test_responses_api_bridge_check_strips_responses_prefix(): """Test that responses_api_bridge_check strips 'responses/' prefix and sets mode.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 4096} model_info, model = responses_api_bridge_check( model="responses/gpt-4-responses", custom_llm_provider="openai", ) assert model == "gpt-4-responses" assert model_info["mode"] == "responses" def test_responses_api_bridge_check_gpt_5_4_pro(): """Test that gpt-5.4-pro routes through responses API bridge, not chat completions. Regression test for https://github.com/BerriAI/litellm/issues/23014 gpt-5.4-pro is a responses-only model and must not be sent to /v1/chat/completions. """ from litellm.main import responses_api_bridge_check for model_name in ["gpt-5.4-pro", "gpt-5.4-pro-2026-03-05"]: model_info, model = responses_api_bridge_check( model=model_name, custom_llm_provider="openai", ) assert ( model_info.get("mode") == "responses" ), f"{model_name} should have mode='responses', got '{model_info.get('mode')}'" def test_responses_api_bridge_check_gpt_5_4_tools_plus_reasoning_routes_to_responses(): """gpt-5.4 with both tools and reasoning_effort should route to Responses API.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.4", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort="xhigh", ) assert model == "gpt-5.4" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_gpt_5_5_tools_plus_reasoning_routes_to_responses(): """gpt-5.5+ with both tools and reasoning_effort should route to Responses API.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.5-pro", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort="xhigh", ) assert model == "gpt-5.5-pro" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_azure_gpt_5_4_tools_plus_reasoning_routes_to_responses(): """Azure gpt-5.4 with both tools and reasoning_effort should route to Responses API.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.4", custom_llm_provider="azure", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort="high", ) assert model == "gpt-5.4" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_azure_gpt_5_4_tools_with_default_reasoning_routes_to_responses(): """ Azure gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables reasoning by default for gpt-5.4+, and Chat Completions rejects function tools whenever reasoning is on. """ from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.4", custom_llm_provider="azure", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort=None, ) assert model == "gpt-5.4" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_gpt_5_4_tools_with_default_reasoning_routes_to_responses(): """ gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables reasoning by default for gpt-5.4+, and Chat Completions rejects function tools whenever reasoning is on ("use /v1/responses or set reasoning_effort to 'none'"). """ from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.4", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort=None, ) assert model == "gpt-5.4" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_gpt_5_4_tools_with_reasoning_none_stays_chat(): """ Explicit reasoning_effort "none" is OpenAI's documented escape hatch that keeps function tools servable on Chat Completions; the bridge must not fire. """ from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.4", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort="none", ) assert model == "gpt-5.4" assert model_info.get("mode") != "responses" def test_responses_api_bridge_check_reasoning_none_with_summary_still_routes_to_responses(): """A reasoning summary is Responses-only regardless of effort value.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.4", custom_llm_provider="openai", reasoning_effort="none", reasoning_summary="detailed", ) assert model == "gpt-5.4" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_gpt_5_4_custom_tools_only_stays_chat(): """ Chat Completions serves custom (grammar) tools natively with reasoning on; only FUNCTION tools trigger the OpenAI rejection. Custom-only requests must stay on chat so responses keep the native custom tool_call shape instead of the bridge's function-shaped mapping. """ from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "custom", "custom": {"name": "ApplyPatch", "description": "V4A patch"}}], reasoning_effort=None, ) assert model == "gpt-5.6" assert model_info.get("mode") != "responses" def test_responses_api_bridge_check_gpt_5_4_mixed_function_and_custom_tools_routes_to_responses(): """One function tool in the mix is enough to make chat unservable with reasoning on.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[ {"type": "custom", "custom": {"name": "ApplyPatch"}}, {"type": "function", "function": {"name": "shell"}}, ], reasoning_effort=None, ) assert model == "gpt-5.6" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_gpt_5_4_flat_function_tool_routes_to_responses(): """Responses-style flat function tool defs still count as function tools.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "name": "shell", "parameters": {"type": "object"}}], reasoning_effort=None, ) assert model == "gpt-5.6" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_dict_effort_none_stays_chat(): """The escape hatch must honor litellm's dict form: {"effort": "none"} means reasoning off.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort={"effort": "none"}, ) assert model == "gpt-5.6" assert model_info.get("mode") != "responses" def test_responses_api_bridge_check_dict_effort_active_routes_to_responses(): from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort={"effort": "low"}, ) assert model == "gpt-5.6" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_dict_effort_none_with_summary_routes_to_responses(): """A summary inside the dict form is Responses-only even when effort is none.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort={"effort": "none", "summary": "concise"}, ) assert model == "gpt-5.6" assert model_info.get("mode") == "responses" @pytest.mark.parametrize("blank_api_base", [None, "", " ", "\t"]) def test_responses_api_bridge_check_blank_api_base_is_default_openai(blank_api_base): """ A blank api_base (None, empty, or whitespace) resolves to the default OpenAI endpoint downstream, which enforces the reasoning+tools constraint, so gpt-5.4+ function-tool requests with unset reasoning_effort must still auto-bridge. """ from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort=None, api_base=blank_api_base, ) assert model == "gpt-5.6" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_custom_api_base_with_unset_effort_stays_chat(): """ Chat-only OpenAI-compatible backends registered under the openai provider with a custom api_base and gpt-5.4+ model names serve tools-without-reasoning fine and have no /responses route; the unset-effort arm must not reroute them. """ from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort=None, api_base="http://vllm.internal:8000/v1", ) assert model == "gpt-5.6" assert model_info.get("mode") != "responses" def test_responses_api_bridge_check_custom_api_base_via_global_with_unset_effort_stays_chat(monkeypatch): """ A custom base set through the litellm.api_base global (not the call arg) is resolved the same way the chat handler resolves it, so the unset-effort arm must not reroute a chat-only backend to a /responses route it lacks. Regression guard: the gate previously inspected only the call-level api_base and bridged these requests. """ import litellm from litellm.main import responses_api_bridge_check monkeypatch.setattr(litellm, "api_base", "http://vllm.internal:8000/v1") with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort=None, api_base=None, ) assert model == "gpt-5.6" assert model_info.get("mode") != "responses" @pytest.mark.parametrize("env_var", ["OPENAI_BASE_URL", "OPENAI_API_BASE"]) def test_responses_api_bridge_check_custom_api_base_via_env_with_unset_effort_stays_chat(monkeypatch, env_var): """ A custom base set via OPENAI_BASE_URL/OPENAI_API_BASE env is resolved identically to the chat handler, so the unset-effort arm leaves the request on chat instead of bridging it. """ import litellm from litellm.main import responses_api_bridge_check monkeypatch.setattr(litellm, "api_base", None) monkeypatch.delenv("OPENAI_BASE_URL", raising=False) monkeypatch.delenv("OPENAI_API_BASE", raising=False) monkeypatch.setenv(env_var, "http://vllm.internal:8000/v1") with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort=None, api_base=None, ) assert model == "gpt-5.6" assert model_info.get("mode") != "responses" def test_responses_api_bridge_check_custom_api_base_with_explicit_effort_still_routes(): """Explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.6", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort="high", api_base="http://vllm.internal:8000/v1", ) assert model == "gpt-5.6" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_azure_with_api_base_and_unset_effort_routes(): """Azure OpenAI always sets api_base and does enforce the constraint; keep bridging.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.4", custom_llm_provider="azure", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort=None, api_base="https://myresource.openai.azure.com", ) assert model == "gpt-5.4" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat(): """Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.1", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort=None, ) assert model == "gpt-5.1" assert model_info.get("mode") != "responses" def test_responses_api_bridge_check_gpt_5_4_reasoning_summary_without_tools_routes_to_responses(): """gpt-5.4+ with reasoning_effort + reasoningSummary but no tools should bridge (AI SDK).""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5.4", custom_llm_provider="openai", tools=None, reasoning_effort="medium", reasoning_summary="auto", ) assert model == "gpt-5.4" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_gpt_5_reasoning_summary_routes_to_responses(): """Bare ``gpt-5`` with reasoning_effort + reasoningSummary should bridge (not 5.4+).""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5", custom_llm_provider="openai", tools=None, reasoning_effort="medium", reasoning_summary="auto", ) assert model == "gpt-5" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_gpt_5_tools_without_summary_stays_chat(): """gpt-5 with tools + reasoning_effort but no summary should stay on chat.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 128000} model_info, model = responses_api_bridge_check( model="gpt-5", custom_llm_provider="openai", tools=[{"type": "function", "function": {"name": "get_capital"}}], reasoning_effort="medium", reasoning_summary=None, ) assert model == "gpt-5" assert model_info.get("mode") != "responses" @patch("litellm.completion_extras.responses_api_bridge.completion") def test_gpt_5_4_responses_bridge_preserves_reasoning_summary_dict( mock_responses_completion, ): """When routed to Responses, preserve reasoning_effort summary dict.""" mock_responses_completion.return_value = MagicMock() import litellm litellm.completion( model="gpt-5.4", messages=[{"role": "user", "content": "What is the capital of France?"}], tools=[ { "type": "function", "function": { "name": "get_capital", "description": "Get the capital of a country", "parameters": { "type": "object", "properties": {"country": {"type": "string"}}, }, }, } ], reasoning_effort={"effort": "xhigh", "summary": "detailed"}, api_key="fake-key", ) assert mock_responses_completion.called is True optional_params = mock_responses_completion.call_args.kwargs["optional_params"] assert optional_params["reasoning_effort"] == { "effort": "xhigh", "summary": "detailed", } @pytest.mark.parametrize( "model, model_info, expected_model_param, expected_base_model_param", [ ("gemini/gemini-3.1-pro", None, "gemini-3.1-pro", None), ( "gemini/gemini-3.1-pro", {"base_model": "gemini-3.1-pro-preview"}, "gemini-3.1-pro", "gemini-3.1-pro-preview", ), ], ) def test_completion_optional_params_base_model( model: str, model_info: dict | None, expected_model_param: str, expected_base_model_param: str | None, ): """``model_info.base_model`` must reach ``get_optional_params`` as ``base_model`` (an additive capability hint), without overwriting ``model`` with the label. Regression for #29618: overwriting ``model`` with a friendly ``base_model`` label made Bedrock drop ``tools``/``tool_choice`` under ``drop_params``.""" with patch("litellm.main.get_optional_params") as mock_get_optional_params: mock_get_optional_params.return_value = MagicMock() import litellm kwargs = { "model": model, "messages": [{"role": "user", "content": "What is the capital of France?"}], "api_key": "fake-key", "mock_response": "Hey, how's it going?", } if model_info is not None: kwargs["model_info"] = model_info litellm.completion(**kwargs) assert mock_get_optional_params.called is True call_kwargs = mock_get_optional_params.call_args.kwargs assert call_kwargs["model"] == expected_model_param assert call_kwargs["base_model"] == expected_base_model_param @patch("litellm.completion_extras.responses_api_bridge.completion") def test_gpt_5_4_responses_bridge_merges_reasoning_summary_kwarg_without_tools( mock_responses_completion, ): """reasoningSummary without tools should route and merge into reasoning_effort dict.""" mock_responses_completion.return_value = MagicMock() import litellm litellm.completion( model="gpt-5.4", messages=[{"role": "user", "content": "ok"}], reasoning_effort="medium", reasoningSummary="auto", api_key="fake-key", ) assert mock_responses_completion.called is True optional_params = mock_responses_completion.call_args.kwargs["optional_params"] assert optional_params["reasoning_effort"] == { "effort": "medium", "summary": "auto", } assert "reasoningSummary" not in optional_params assert "reasoning_summary" not in optional_params @patch("litellm.completion_extras.responses_api_bridge.completion") def test_responses_bridge_preserves_reasoning_summary_without_effort( mock_responses_completion, ): """Reasoning summary should survive responses routing even without effort.""" mock_responses_completion.return_value = MagicMock() import litellm with patch.object(litellm, "route_all_chat_openai_to_responses", True): litellm.completion( model="gpt-4o", messages=[{"role": "user", "content": "ok"}], reasoningSummary="auto", api_key="fake-key", ) assert mock_responses_completion.called is True optional_params = mock_responses_completion.call_args.kwargs["optional_params"] assert optional_params["reasoning_effort"] == {"summary": "auto"} assert "reasoningSummary" not in optional_params assert "reasoning_summary" not in optional_params @patch("litellm.completion_extras.responses_api_bridge.completion") def test_gpt_5_responses_bridge_tools_and_reasoning_summary( mock_responses_completion, ): """Bare gpt-5 with tools + reasoningSummary should bridge (OpenCode-style).""" mock_responses_completion.return_value = MagicMock() import litellm litellm.completion( model="gpt-5", messages=[{"role": "user", "content": "ok"}], tools=[ { "type": "function", "function": { "name": "apply_patch", "parameters": {"type": "object", "properties": {}}, }, } ], tool_choice="auto", reasoning_effort="medium", reasoningSummary="auto", stream=True, api_key="fake-key", ) assert mock_responses_completion.called is True optional_params = mock_responses_completion.call_args.kwargs["optional_params"] assert optional_params.get("reasoning_effort") == { "effort": "medium", "summary": "auto", } def test_responses_api_bridge_check_handles_exception(): """Test that responses_api_bridge_check handles exceptions and still processes responses/ models.""" from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.side_effect = Exception("Model not found") model_info, model = responses_api_bridge_check( model="responses/custom-model", custom_llm_provider="custom" ) assert model == "custom-model" assert model_info["mode"] == "responses" def test_responses_api_bridge_check_global_flag_routes_openai(): """When route_all_chat_openai_to_responses is True, any OpenAI model routes to responses.""" from litellm.main import responses_api_bridge_check with patch.object(litellm, "route_all_chat_openai_to_responses", True): model_info, model = responses_api_bridge_check( model="gpt-4o", custom_llm_provider="openai", ) assert model == "gpt-4o" assert model_info.get("mode") == "responses" def test_responses_api_bridge_check_global_flag_does_not_affect_azure(): """route_all_chat_openai_to_responses should not affect Azure models.""" from litellm.main import responses_api_bridge_check with patch.object(litellm, "route_all_chat_openai_to_responses", True): with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 4096} model_info, model = responses_api_bridge_check( model="gpt-4o", custom_llm_provider="azure", ) assert model_info.get("mode") != "responses" def test_responses_api_bridge_check_global_flag_default_false(): """By default, route_all_chat_openai_to_responses is False and doesn't affect routing.""" from litellm.main import responses_api_bridge_check with patch.object(litellm, "route_all_chat_openai_to_responses", False): with patch("litellm.main._get_model_info_helper") as mock_get_model_info: mock_get_model_info.return_value = {"max_tokens": 4096} model_info, model = responses_api_bridge_check( model="gpt-4o", custom_llm_provider="openai", ) assert model_info.get("mode") != "responses" @pytest.mark.asyncio async def test_async_mock_delay(): """Use asyncio await for mock delay on acompletion""" import time from litellm import acompletion start_time = time.time() result = await acompletion( model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hey, how's it going?"}], mock_delay=0.01, mock_response="Hello world", ) end_time = time.time() delay = end_time - start_time assert delay >= 0.01 def test_stream_chunk_builder_thinking_blocks(): from litellm import stream_chunk_builder from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices chunks = [ ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=0, delta=Delta( reasoning_content="I need to summar", thinking_blocks=[ { "type": "thinking", "thinking": "I need to summar", "signature": None, } ], provider_specific_fields={ "thinking_blocks": [ { "type": "thinking", "thinking": "I need to summar", "signature": None, } ] }, content="", role="assistant", function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=0, delta=Delta( reasoning_content="ize the previous agent's thinking process into a", thinking_blocks=[ { "type": "thinking", "thinking": "ize the previous agent's thinking process into a", "signature": None, } ], provider_specific_fields={ "thinking_blocks": [ { "type": "thinking", "thinking": "ize the previous agent's thinking process into a", "signature": None, } ] }, content="", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=0, delta=Delta( reasoning_content=" short description. Based on the input data provide", thinking_blocks=[ { "type": "thinking", "thinking": " short description. Based on the input data provide", "signature": None, } ], provider_specific_fields={ "thinking_blocks": [ { "type": "thinking", "thinking": " short description. Based on the input data provide", "signature": None, } ] }, content="", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=0, delta=Delta( reasoning_content="d, it seems the agent was planning to refine their search", thinking_blocks=[ { "type": "thinking", "thinking": "d, it seems the agent was planning to refine their search", "signature": None, } ], provider_specific_fields={ "thinking_blocks": [ { "type": "thinking", "thinking": "d, it seems the agent was planning to refine their search", "signature": None, } ] }, content="", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=0, delta=Delta( reasoning_content=" to focus more on technical aspects of home automation and home", thinking_blocks=[ { "type": "thinking", "thinking": " to focus more on technical aspects of home automation and home", "signature": None, } ], provider_specific_fields={ "thinking_blocks": [ { "type": "thinking", "thinking": " to focus more on technical aspects of home automation and home", "signature": None, } ] }, content="", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=0, delta=Delta( reasoning_content=" energy system management.\n\nI'll create a brief", thinking_blocks=[ { "type": "thinking", "thinking": " energy system management.\n\nI'll create a brief", "signature": None, } ], provider_specific_fields={ "thinking_blocks": [ { "type": "thinking", "thinking": " energy system management.\n\nI'll create a brief", "signature": None, } ] }, content="", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=0, delta=Delta( reasoning_content=" summary of what the agent was doing.", thinking_blocks=[ { "type": "thinking", "thinking": " summary of what the agent was doing.", "signature": None, } ], provider_specific_fields={ "thinking_blocks": [ { "type": "thinking", "thinking": " summary of what the agent was doing.", "signature": None, } ] }, content="", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=0, delta=Delta( reasoning_content="", thinking_blocks=[ { "type": "thinking", "thinking": "", "signature": "ErUBCkYIBRgCIkAKBSMkB2+MBF643wiWxlERsGXVdlhbPx9lnTIbygzjFIeZ5uhTV+HNWDon9vQV4hmXvAKwQfwS8vkNFB366l05Egzt2U18IpRrZRyQn1UaDDdYvKHYP8Ps1IbWjSIw8eSYOU9gtqNcwR6D0wY7iOPx2GliDEatLI5rSs96CByoTIoADL2M5bX8KP0jEpbHKh0ccYryigdH/3J8EiFt/BmGUceVASP5l9r22dFWiBgC", } ], provider_specific_fields={ "thinking_blocks": [ { "type": "thinking", "thinking": "", "signature": "ErUBCkYIBRgCIkAKBSMkB2+MBF643wiWxlERsGXVdlhbPx9lnTIbygzjFIeZ5uhTV+HNWDon9vQV4hmXvAKwQfwS8vkNFB366l05Egzt2U18IpRrZRyQn1UaDDdYvKHYP8Ps1IbWjSIw8eSYOU9gtqNcwR6D0wY7iOPx2GliDEatLI5rSs96CByoTIoADL2M5bX8KP0jEpbHKh0ccYryigdH/3J8EiFt/BmGUceVASP5l9r22dFWiBgC", } ] }, content="", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=1, delta=Delta( provider_specific_fields=None, content='{"a', role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=1, delta=Delta( provider_specific_fields=None, content='gent_doing"', role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=1, delta=Delta( provider_specific_fields=None, content=': "Re', role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=1, delta=Delta( provider_specific_fields=None, content="searching", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=1, delta=Delta( provider_specific_fields=None, content=" technic", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=1, delta=Delta( provider_specific_fields=None, content="al aspect", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=1, delta=Delta( provider_specific_fields=None, content="s of home au", role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason=None, index=1, delta=Delta( provider_specific_fields=None, content='tomation"}', role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, citations=None, ), ModelResponseStream( id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9", created=1751934860, model="claude-3-7-sonnet-latest", object="chat.completion.chunk", system_fingerprint=None, choices=[ StreamingChoices( finish_reason="tool_calls", index=0, delta=Delta( provider_specific_fields=None, content=None, role=None, function_call=None, tool_calls=None, audio=None, ), logprobs=None, ) ], provider_specific_fields=None, ), ] response = stream_chunk_builder(chunks=chunks) print(response) assert response is not None assert response.choices[0].message.content is not None assert response.choices[0].message.thinking_blocks is not None from litellm.llms.openai.openai import OpenAIChatCompletion def throw_retryable_error(*_, **__): raise RuntimeError("BOOM") @pytest.mark.asyncio async def test_retrying() -> None: litellm.num_retries = 10 with ( patch.object( OpenAIChatCompletion, "make_openai_chat_completion_request", side_effect=throw_retryable_error, ) as mock_request, pytest.raises(litellm.InternalServerError, match="LiteLLM Retried: 10 times"), ): await litellm.acompletion( model="gpt-4o-mini", messages=[{"role": "user", "content": "Hello"}], ) def test_anthropic_disable_url_suffix_env_var(): """Test that LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX prevents /v1/messages suffix.""" import os from unittest.mock import MagicMock, patch from litellm import completion # Test with environment variable disabled (default behavior) with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}): actual_api_base = None with patch("litellm.main.anthropic_chat_completions") as mock_anthropic: def capture_completion(**kwargs): nonlocal actual_api_base actual_api_base = kwargs.get("api_base") mock_response = MagicMock() mock_response.choices = [MagicMock()] return mock_response mock_anthropic.completion = capture_completion # This should append /v1/messages completion( model="anthropic/claude-3-sonnet", messages=[{"role": "user", "content": "test"}], api_key="test-key", ) # Verify the api_base has /v1/messages appended assert actual_api_base.endswith("/v1/messages") assert actual_api_base == "https://api.example.com/v1/messages" # Test with environment variable enabled with patch.dict( os.environ, { "ANTHROPIC_API_BASE": "https://api.example.com/custom/path", "LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX": "true", }, ): actual_api_base = None with patch("litellm.main.anthropic_chat_completions") as mock_anthropic: def capture_completion(**kwargs): nonlocal actual_api_base actual_api_base = kwargs.get("api_base") mock_response = MagicMock() mock_response.choices = [MagicMock()] return mock_response mock_anthropic.completion = capture_completion # This should NOT append /v1/messages completion( model="anthropic/claude-3-sonnet", messages=[{"role": "user", "content": "test"}], api_key="test-key", ) # Verify the api_base does not have /v1/messages appended assert actual_api_base == "https://api.example.com/custom/path" assert not actual_api_base.endswith("/v1/messages") def test_anthropic_text_disable_url_suffix_env_var(): """Test that LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX prevents /v1/complete suffix for anthropic_text.""" import os from unittest.mock import MagicMock, patch from litellm import completion # Test with environment variable disabled (default behavior) with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}): actual_api_base = None with patch("litellm.main.base_llm_http_handler") as mock_handler: def capture_completion(**kwargs): nonlocal actual_api_base actual_api_base = kwargs.get("api_base") return MagicMock() mock_handler.completion = capture_completion # This should append /v1/complete completion( model="anthropic_text/claude-instant-1", messages=[{"role": "user", "content": "test"}], api_key="test-key", ) # Verify the api_base has /v1/complete appended assert actual_api_base.endswith("/v1/complete") assert actual_api_base == "https://api.example.com/v1/complete" # Test with environment variable enabled with patch.dict( os.environ, { "ANTHROPIC_API_BASE": "https://api.example.com/custom/complete", "LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX": "true", }, ): actual_api_base = None with patch("litellm.main.base_llm_http_handler") as mock_handler: def capture_completion(**kwargs): nonlocal actual_api_base actual_api_base = kwargs.get("api_base") return MagicMock() mock_handler.completion = capture_completion # This should NOT append /v1/complete completion( model="anthropic_text/claude-instant-1", messages=[{"role": "user", "content": "test"}], api_key="test-key", ) # Verify the api_base does not have /v1/complete appended assert actual_api_base == "https://api.example.com/custom/complete" assert not actual_api_base.endswith("/v1/complete") def test_image_edit_merges_headers_and_extra_headers(): from litellm.images.main import base_llm_http_handler combined_headers = { "x-test-header-one": "value-1", "x-test-header-two": "value-2", } mock_image_edit_config = MagicMock() mock_image_edit_config.get_supported_openai_params.return_value = set() mock_image_edit_config.map_openai_params.side_effect = lambda **kwargs: dict( kwargs["image_edit_optional_params"] ) with ( patch( "litellm.images.main.ProviderConfigManager.get_provider_image_edit_config", return_value=mock_image_edit_config, ) as mock_config, patch.object( base_llm_http_handler, "image_edit_handler", return_value="ok", ) as mock_handler, ): response = litellm.image_edit( image=MagicMock(name="image"), prompt="test", model="azure/gpt-image-1", headers={"x-test-header-one": "value-1"}, extra_headers={ "x-test-header-two": "value-2", }, ) assert response == "ok" mock_config.assert_called_once() handler_kwargs = mock_handler.call_args.kwargs assert handler_kwargs["extra_headers"] == combined_headers assert "extra_headers" not in handler_kwargs["image_edit_optional_request_params"] def test_mock_completion_stream_with_model_response(): """Test that mock_completion correctly handles stream=True with a ModelResponse as mock_response.""" from litellm import completion from litellm.types.utils import Choices, Message, ModelResponse, Usage # Create a ModelResponse object mock_model_response = ModelResponse( id="chatcmpl-test-123", created=1234567890, model="gpt-4o-mini", object="chat.completion", choices=[ Choices( finish_reason="stop", index=0, message=Message( content="This is a test response", role="assistant", ), ) ], usage=Usage( prompt_tokens=10, completion_tokens=20, total_tokens=30, ), ) # Call completion with stream=True and mock_response as ModelResponse response = completion( model="gpt-4o-mini", messages=[{"role": "user", "content": "Hello"}], stream=True, mock_response=mock_model_response, ) # Verify that the response is a stream assert response is not None # Collect all chunks from the stream chunks = [] for chunk in response: chunks.append(chunk) print(f"Chunk: {chunk}") # Verify we got chunks assert len(chunks) > 0 # Verify the content is streamed correctly accumulated_content = "" for chunk in chunks: if ( hasattr(chunk.choices[0].delta, "content") and chunk.choices[0].delta.content ): accumulated_content += chunk.choices[0].delta.content assert "This is a test response" in accumulated_content or len(chunks) > 0 @pytest.mark.asyncio async def test_async_mock_completion_stream_with_model_response(): """Test that async mock_completion correctly handles stream=True with a ModelResponse as mock_response.""" from litellm import acompletion from litellm.types.utils import Choices, Message, ModelResponse, Usage # Create a ModelResponse object mock_model_response = ModelResponse( id="chatcmpl-test-456", created=1234567890, model="gpt-4o-mini", object="chat.completion", choices=[ Choices( finish_reason="stop", index=0, message=Message( content="This is an async test response", role="assistant", ), ) ], usage=Usage( prompt_tokens=15, completion_tokens=25, total_tokens=40, ), ) # Call acompletion with stream=True and mock_response as ModelResponse response = await acompletion( model="gpt-4o-mini", messages=[{"role": "user", "content": "Hello async"}], stream=True, mock_response=mock_model_response, ) # Verify that the response is a stream assert response is not None # Collect all chunks from the stream chunks = [] async for chunk in response: chunks.append(chunk) print(f"Async Chunk: {chunk}") # Verify we got chunks assert len(chunks) > 0 # Verify the content is streamed correctly accumulated_content = "" for chunk in chunks: if ( hasattr(chunk.choices[0].delta, "content") and chunk.choices[0].delta.content ): accumulated_content += chunk.choices[0].delta.content assert "This is an async test response" in accumulated_content or len(chunks) > 0 class TestCallTypesOCR: """Test that OCR call types are properly defined in CallTypes enum. Fixes https://github.com/BerriAI/litellm/issues/17381 """ def test_ocr_call_type_exists(self): """Test that CallTypes.ocr exists and has correct value.""" from litellm.types.utils import CallTypes assert hasattr(CallTypes, "ocr") assert CallTypes.ocr.value == "ocr" def test_aocr_call_type_exists(self): """Test that CallTypes.aocr exists and has correct value.""" from litellm.types.utils import CallTypes assert hasattr(CallTypes, "aocr") assert CallTypes.aocr.value == "aocr" def test_ocr_call_type_from_string(self): """Test that CallTypes can be constructed from 'ocr' string.""" from litellm.types.utils import CallTypes call_type = CallTypes("ocr") assert call_type == CallTypes.ocr def test_aocr_call_type_from_string(self): """Test that CallTypes can be constructed from 'aocr' string. This is the actual use case that was failing - the OCR endpoint uses route_type='aocr' and guardrails try to instantiate CallTypes('aocr'). """ from litellm.types.utils import CallTypes call_type = CallTypes("aocr") assert call_type == CallTypes.aocr def test_stream_chunk_builder_text_completion_combines_text_and_usage(): from litellm.main import stream_chunk_builder_text_completion from litellm.types.utils import TextCompletionResponse chunks = [ TextCompletionResponse( id="cmpl-1", object="text_completion", created=1, model="gpt-3.5-turbo-instruct", choices=[{"text": "Hello", "index": 0, "logprobs": None, "finish_reason": None}], ), TextCompletionResponse( id="cmpl-1", object="text_completion", created=1, model="gpt-3.5-turbo-instruct", choices=[{"text": " world", "index": 0, "logprobs": None, "finish_reason": "stop"}], ), ] response = stream_chunk_builder_text_completion( chunks=chunks, messages=[{"role": "user", "content": "say hello"}] ) assert response.choices[0].text == "Hello world" assert response.choices[0].finish_reason == "stop" assert response.usage.prompt_tokens > 0 assert response.usage.completion_tokens > 0 assert response.usage.total_tokens == response.usage.prompt_tokens + response.usage.completion_tokens @pytest.mark.asyncio @pytest.mark.parametrize( "aws_credential_kwargs", [ { "aws_session_name": "litellm-gcp", "aws_role_name": "arn:aws:iam::123456789012:role/litellm-bedrock-role", "aws_web_identity_token": "oidc/google/108963886734710037768", }, { "aws_access_key_id": "AKIASTATICKEYFORTEST", "aws_secret_access_key": "static-secret-key", "aws_session_token": "static-session-token", }, ], ids=["web_identity", "static_keys"], ) async def test_acompletion_forwards_aws_credentials_through_responses_bridge( respx_mock: respx.MockRouter, monkeypatch, aws_credential_kwargs: dict ): from botocore.credentials import Credentials from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM original_disable_aiohttp = litellm.disable_aiohttp_transport try: litellm.disable_aiohttp_transport = True monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") litellm.in_memory_llm_clients_cache.flush_cache() monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False) monkeypatch.delenv("BEDROCK_MANTLE_API_KEY", raising=False) get_credentials_mock = MagicMock(return_value=Credentials("fake-key", "fake-secret")) monkeypatch.setattr(BaseAWSLLM, "get_credentials", get_credentials_mock) respx_mock.post("https://bedrock-mantle.us-east-2.api.aws/openai/v1/responses").respond( json={ "id": "resp_123", "object": "response", "created_at": 1760144904, "status": "completed", "model": "openai.gpt-5.4", "output": [ { "type": "message", "id": "msg_1", "role": "assistant", "status": "completed", "content": [{"type": "output_text", "text": "ok", "annotations": []}], } ], } ) response = await litellm.acompletion( model="bedrock_mantle/openai.gpt-5.4", messages=[{"role": "user", "content": "hi"}], api_base="https://bedrock-mantle.us-east-2.api.aws/v1", aws_region_name="us-east-2", num_retries=0, **aws_credential_kwargs, ) assert response.choices[0].message.content == "ok" credential_kwargs = get_credentials_mock.call_args.kwargs assert credential_kwargs["aws_region_name"] == "us-east-2" for key, value in aws_credential_kwargs.items(): assert credential_kwargs[key] == value authorization = respx_mock.calls.last.request.headers["Authorization"] assert authorization.startswith("AWS4-HMAC-SHA256") assert "fake-key" in authorization finally: litellm.disable_aiohttp_transport = original_disable_aiohttp litellm.in_memory_llm_clients_cache.flush_cache() _GEMINI_RESPONSE_BODY = { "candidates": [{"content": {"parts": [{"text": "hello"}], "role": "model"}, "finishReason": "STOP"}], "usageMetadata": {"promptTokenCount": 2, "candidatesTokenCount": 1, "totalTokenCount": 3}, } def _gemini_client_returning_a_reply(): """An injected HTTP client whose post() answers like generativelanguage does.""" from litellm.llms.custom_httpx.http_handler import HTTPHandler client = HTTPHandler() request = httpx.Request("POST", "https://generativelanguage.googleapis.com/") post = MagicMock(return_value=httpx.Response(200, json=_GEMINI_RESPONSE_BODY, request=request)) return client, post @pytest.fixture def restore_model_registry(): """litellm.model_cost and the provider name sets are module-global. register_model merges into the existing entry in place, hence the deep copy. """ model_cost = copy.deepcopy(litellm.model_cost) openai_models = set(litellm.open_ai_chat_completion_models) yield litellm.model_cost.clear() litellm.model_cost.update(model_cost) litellm.open_ai_chat_completion_models.clear() litellm.open_ai_chat_completion_models.update(openai_models) def test_openai_model_name_does_not_outrank_explicit_provider(): """`gemini/gpt-4o` goes to Google, not to litellm's OpenAI handler. completion() checks `model in litellm.open_ai_chat_completion_models` ahead of the gemini branch, so the call used to reach the OpenAI handler carrying VertexGeminiConfig, whose transform_request raises NotImplementedError. """ assert "gpt-4o" in litellm.open_ai_chat_completion_models client, post = _gemini_client_returning_a_reply() with patch.object(client, "post", new=post): response = litellm.completion( model="gemini/gpt-4o", messages=[{"role": "user", "content": "hello"}], api_key="test-api-key", client=client, ) assert "generativelanguage.googleapis.com" in post.call_args.kwargs["url"] assert "models/gpt-4o" in post.call_args.kwargs["url"] assert response.choices[0].message.content == "hello" def test_mislabelled_pricing_entry_does_not_reroute_provider(restore_model_registry): """register_model is the other way into the same failure. An entry claiming litellm_provider "openai" adds its name to open_ai_chat_completion_models, so one mislabelled price reroutes every later call to that model in the process. """ litellm.register_model( { "gemini-2.5-pro": { "litellm_provider": "openai", "mode": "chat", "input_cost_per_token": 1e-06, "output_cost_per_token": 4e-06, } } ) assert "gemini-2.5-pro" in litellm.open_ai_chat_completion_models client, post = _gemini_client_returning_a_reply() with patch.object(client, "post", new=post): response = litellm.completion( model="gemini/gemini-2.5-pro", messages=[{"role": "user", "content": "hello"}], api_key="test-api-key", client=client, ) assert "generativelanguage.googleapis.com" in post.call_args.kwargs["url"] assert response.choices[0].message.content == "hello" def test_openai_model_without_a_provider_still_routes_to_openai(): from openai import OpenAI client = OpenAI(api_key="fake-key") raw_response = client.chat.completions.with_raw_response with patch.object(raw_response, "create") as mock_create, contextlib.suppress(Exception): litellm.completion( model="gpt-4o", messages=[{"role": "user", "content": "hello"}], client=client, ) mock_create.assert_called()