mirror of
https://github.com/BerriAI/litellm.git
synced 2026-08-28 05:25:59 +00:00
3073 lines
112 KiB
Python
3073 lines
112 KiB
Python
import asyncio
|
|
import base64
|
|
import contextlib
|
|
import copy
|
|
import json
|
|
import os
|
|
from collections.abc import Mapping
|
|
from dataclasses import dataclass
|
|
from typing import Final
|
|
|
|
import httpx
|
|
import pytest
|
|
import respx
|
|
from fastapi.testclient import TestClient
|
|
|
|
|
|
import urllib.parse
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
import litellm
|
|
from litellm import main as litellm_main
|
|
from litellm.integrations.custom_logger import CustomLogger
|
|
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs
|
|
from litellm.types.utils import Usage
|
|
|
|
|
|
async def _async_fake_bedrock_image_details(image_url):
|
|
return "ZmFrZS1pbWFnZQ==", "image/png"
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def clear_client_cache():
|
|
"""
|
|
Clear the HTTP client cache before each test to ensure mocks are used.
|
|
This prevents cached real clients from being reused across tests.
|
|
"""
|
|
cache = getattr(litellm, "in_memory_llm_clients_cache", None)
|
|
if cache is not None:
|
|
cache.flush_cache()
|
|
yield
|
|
if cache is not None:
|
|
cache.flush_cache()
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def add_api_keys_to_env(monkeypatch):
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-api03-1234567890")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-api03-1234567890")
|
|
monkeypatch.setenv("AWS_ACCESS_KEY_ID", "my-fake-aws-access-key-id")
|
|
monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "my-fake-aws-secret-access-key")
|
|
monkeypatch.setenv("AWS_REGION", "us-east-1")
|
|
# Keep these transformation tests on the simple access-key path. A leaked
|
|
# session token or role/web-identity env var pushes Bedrock auth down a
|
|
# different branch and fails before the mocked HTTP client is exercised.
|
|
monkeypatch.delenv("AWS_SESSION_TOKEN", raising=False)
|
|
monkeypatch.delenv("AWS_ROLE_ARN", raising=False)
|
|
monkeypatch.delenv("AWS_WEB_IDENTITY_TOKEN_FILE", raising=False)
|
|
|
|
|
|
@pytest.fixture
|
|
def openai_api_response():
|
|
mock_response_data = {
|
|
"id": "chatcmpl-B0W3vmiM78Xkgx7kI7dr7PC949DMS",
|
|
"choices": [
|
|
{
|
|
"finish_reason": "stop",
|
|
"index": 0,
|
|
"logprobs": None,
|
|
"message": {
|
|
"content": "",
|
|
"refusal": None,
|
|
"role": "assistant",
|
|
"audio": None,
|
|
"function_call": None,
|
|
"tool_calls": None,
|
|
},
|
|
}
|
|
],
|
|
"created": 1739462947,
|
|
"model": "gpt-4o-mini-2024-07-18",
|
|
"object": "chat.completion",
|
|
"service_tier": "default",
|
|
"system_fingerprint": "fp_bd83329f63",
|
|
"usage": {
|
|
"completion_tokens": 1,
|
|
"prompt_tokens": 121,
|
|
"total_tokens": 122,
|
|
"completion_tokens_details": {
|
|
"accepted_prediction_tokens": 0,
|
|
"audio_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"rejected_prediction_tokens": 0,
|
|
},
|
|
"prompt_tokens_details": {"audio_tokens": 0, "cached_tokens": 0},
|
|
},
|
|
}
|
|
|
|
return mock_response_data
|
|
|
|
|
|
def test_completion_missing_role(openai_api_response):
|
|
from openai import OpenAI
|
|
|
|
from litellm.types.utils import ModelResponse
|
|
|
|
client = OpenAI(api_key="test_api_key")
|
|
|
|
mock_raw_response = MagicMock()
|
|
mock_raw_response.headers = {
|
|
"x-request-id": "123",
|
|
"openai-organization": "org-123",
|
|
"x-ratelimit-limit-requests": "100",
|
|
"x-ratelimit-remaining-requests": "99",
|
|
}
|
|
mock_raw_response.parse.return_value = ModelResponse(**openai_api_response)
|
|
|
|
print(f"openai_api_response: {openai_api_response}")
|
|
|
|
with patch.object(
|
|
client.chat.completions.with_raw_response, "create", mock_raw_response
|
|
) as mock_create:
|
|
litellm.completion(
|
|
model="gpt-4o-mini",
|
|
messages=[
|
|
{"role": "user", "content": "Hey"},
|
|
{
|
|
"content": "",
|
|
"tool_calls": [
|
|
{
|
|
"id": "call_m0vFJjQmTH1McvaHBPR2YFwY",
|
|
"function": {
|
|
"arguments": '{"input": "dksjsdkjdhskdjshdskhjkhlk"}',
|
|
"name": "tool_name",
|
|
},
|
|
"type": "function",
|
|
"index": 0,
|
|
},
|
|
{
|
|
"id": "call_Vw6RaqV2n5aaANXEdp5pYxo2",
|
|
"function": {
|
|
"arguments": '{"input": "jkljlkjlkjlkjlk"}',
|
|
"name": "tool_name",
|
|
},
|
|
"type": "function",
|
|
"index": 1,
|
|
},
|
|
{
|
|
"id": "call_hBIKwldUEGlNh6NlSXil62K4",
|
|
"function": {
|
|
"arguments": '{"input": "jkjlkjlkjlkj;lj"}',
|
|
"name": "tool_name",
|
|
},
|
|
"type": "function",
|
|
"index": 2,
|
|
},
|
|
],
|
|
},
|
|
],
|
|
client=client,
|
|
)
|
|
|
|
mock_create.assert_called_once()
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"model",
|
|
[
|
|
"gemini/gemini-1.5-flash",
|
|
"bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
"bedrock/invoke/anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
"anthropic/claude-3-5-sonnet",
|
|
],
|
|
)
|
|
@pytest.mark.parametrize("sync_mode", [True, False])
|
|
@pytest.mark.asyncio
|
|
async def test_url_with_format_param(model, sync_mode, monkeypatch):
|
|
from litellm import acompletion, completion
|
|
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
|
from litellm.litellm_core_utils.prompt_templates import factory as prompt_factory
|
|
|
|
if sync_mode:
|
|
client = HTTPHandler()
|
|
else:
|
|
client = AsyncHTTPHandler()
|
|
|
|
# This test is about request shaping, not live image downloads. Stub the
|
|
# URL->image conversion helpers so suite-level network/client state from
|
|
# earlier tests cannot prevent the mocked provider client from being hit.
|
|
fake_base64_image = "data:image/png;base64,ZmFrZS1pbWFnZQ=="
|
|
monkeypatch.setattr(
|
|
prompt_factory, "convert_url_to_base64", lambda url: fake_base64_image
|
|
)
|
|
monkeypatch.setattr(
|
|
prompt_factory.BedrockImageProcessor,
|
|
"get_image_details",
|
|
staticmethod(lambda image_url: ("ZmFrZS1pbWFnZQ==", "image/png")),
|
|
)
|
|
monkeypatch.setattr(
|
|
prompt_factory.BedrockImageProcessor,
|
|
"get_image_details_async",
|
|
staticmethod(_async_fake_bedrock_image_details),
|
|
)
|
|
|
|
args = {
|
|
"model": model,
|
|
"messages": [
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{
|
|
"type": "image_url",
|
|
"image_url": {
|
|
"url": "https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/c233c9ade2ccb5491072ae232c814942.png",
|
|
"format": "image/png",
|
|
},
|
|
},
|
|
{"type": "text", "text": "Describe this image"},
|
|
],
|
|
}
|
|
],
|
|
}
|
|
if model.startswith("gemini/"):
|
|
args["api_key"] = "test-api-key"
|
|
with patch.object(client, "post", new=MagicMock()) as mock_client:
|
|
try:
|
|
if sync_mode:
|
|
response = completion(**args, client=client)
|
|
else:
|
|
response = await acompletion(**args, client=client)
|
|
print(response)
|
|
except Exception as e:
|
|
pass
|
|
|
|
mock_client.assert_called()
|
|
|
|
print(mock_client.call_args.kwargs)
|
|
|
|
if "data" in mock_client.call_args.kwargs:
|
|
json_str = mock_client.call_args.kwargs["data"]
|
|
else:
|
|
json_str = json.dumps(mock_client.call_args.kwargs["json"])
|
|
|
|
if isinstance(json_str, bytes):
|
|
json_str = json_str.decode("utf-8")
|
|
|
|
print(f"type of json_str: {type(json_str)}")
|
|
|
|
# Bedrock models convert URLs to base64, while direct Anthropic models support URLs
|
|
# bedrock/invoke models use Anthropic messages API which supports URLs
|
|
if model.startswith("bedrock/invoke/"):
|
|
# bedrock/invoke should convert URLs to base64 (doesn't support URL references)
|
|
# URL should NOT be in the JSON (it should be converted to base64)
|
|
assert "https://awsmp-logos.s3.amazonaws.com" not in json_str
|
|
# Should have base64 data in the source (type="base64", not type="url")
|
|
assert '"type":"base64"' in json_str or '"type": "base64"' in json_str
|
|
# Should have "data" field containing base64 content
|
|
assert '"data"' in json_str
|
|
elif model.startswith("bedrock/"):
|
|
# Regular Bedrock models should convert URLs to base64 (uses "bytes" field)
|
|
# URL should NOT be in the JSON (it should be converted to base64)
|
|
assert "https://awsmp-logos.s3.amazonaws.com" not in json_str
|
|
# Should have "bytes" field (Bedrock uses "bytes" not "base64" in the field name)
|
|
assert '"bytes"' in json_str or '"bytes":' in json_str
|
|
elif model.startswith("anthropic/"):
|
|
# Direct Anthropic models should pass HTTPS URLs directly (HTTP URLs are converted to base64)
|
|
# Since we're using HTTPS URL, it should be passed as-is
|
|
assert "https://awsmp-logos.s3.amazonaws.com" in json_str
|
|
# For Anthropic, URL references use "url" type, not base64
|
|
assert '"type":"url"' in json_str or '"type": "url"' in json_str
|
|
else:
|
|
# For other models, check format parameter is respected
|
|
assert "png" in json_str
|
|
assert "jpeg" not in json_str
|
|
|
|
|
|
@pytest.mark.parametrize("model", ["gpt-4o-mini"])
|
|
@pytest.mark.parametrize("sync_mode", [True, False])
|
|
@pytest.mark.asyncio
|
|
async def test_url_with_format_param_openai(model, sync_mode):
|
|
from openai import AsyncOpenAI, OpenAI
|
|
|
|
from litellm import acompletion, completion
|
|
|
|
if sync_mode:
|
|
client = OpenAI()
|
|
else:
|
|
client = AsyncOpenAI()
|
|
|
|
args = {
|
|
"model": model,
|
|
"messages": [
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{
|
|
"type": "image_url",
|
|
"image_url": {
|
|
"url": "https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/c233c9ade2ccb5491072ae232c814942.png",
|
|
"format": "image/png",
|
|
},
|
|
},
|
|
{"type": "text", "text": "Describe this image"},
|
|
],
|
|
}
|
|
],
|
|
}
|
|
with patch.object(
|
|
client.chat.completions.with_raw_response, "create"
|
|
) as mock_client:
|
|
try:
|
|
if sync_mode:
|
|
response = completion(**args, client=client)
|
|
else:
|
|
response = await acompletion(**args, client=client)
|
|
print(response)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_client.assert_called()
|
|
|
|
print(mock_client.call_args.kwargs)
|
|
|
|
json_str = json.dumps(mock_client.call_args.kwargs)
|
|
|
|
assert "format" not in json_str
|
|
|
|
|
|
def test_bedrock_latency_optimized_inference():
|
|
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
|
|
|
client = HTTPHandler()
|
|
with patch.object(client, "post") as mock_post:
|
|
try:
|
|
response = litellm.completion(
|
|
model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
|
performanceConfig={"latency": "optimized"},
|
|
client=client,
|
|
)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_post.assert_called_once()
|
|
json_data = json.loads(mock_post.call_args.kwargs["data"])
|
|
assert json_data["performanceConfig"]["latency"] == "optimized"
|
|
|
|
|
|
def test_strip_input_examples_for_non_anthropic_providers():
|
|
tools = [
|
|
{
|
|
"type": "function",
|
|
"name": "example_tool",
|
|
"input_examples": [{"foo": "bar"}],
|
|
"function": {
|
|
"name": "example_tool",
|
|
"input_examples": [{"foo": "bar"}],
|
|
},
|
|
}
|
|
]
|
|
|
|
assert not litellm_main._should_allow_input_examples(
|
|
custom_llm_provider="openai", model="gpt-4o-mini"
|
|
)
|
|
|
|
cleaned = litellm_main._drop_input_examples_from_tools(tools=tools)
|
|
|
|
assert isinstance(cleaned, list)
|
|
assert "input_examples" not in cleaned[0]
|
|
assert "input_examples" not in cleaned[0]["function"]
|
|
|
|
|
|
def test_custom_provider_with_extra_headers():
|
|
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
|
|
|
with patch.object(
|
|
litellm.llms.custom_httpx.http_handler.HTTPHandler, "post"
|
|
) as mock_post:
|
|
response = litellm.completion(
|
|
model="custom/custom",
|
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
|
headers={"X-Custom-Header": "custom-value"},
|
|
api_base="https://example.com/api/v1",
|
|
)
|
|
|
|
mock_post.assert_called_once()
|
|
assert mock_post.call_args[1]["headers"]["X-Custom-Header"] == "custom-value"
|
|
|
|
|
|
def test_custom_provider_with_extra_body():
|
|
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
|
|
|
with patch.object(
|
|
litellm.llms.custom_httpx.http_handler.HTTPHandler, "post"
|
|
) as mock_post:
|
|
response = litellm.completion(
|
|
model="custom/custom",
|
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
|
extra_body={
|
|
"X-Custom-BodyValue": "custom-value",
|
|
"X-Custom-BodyValue2": "custom-value2",
|
|
},
|
|
api_base="https://example.com/api/v1",
|
|
)
|
|
mock_post.assert_called_once()
|
|
|
|
assert mock_post.call_args[1]["json"]["X-Custom-BodyValue"] == "custom-value"
|
|
assert mock_post.call_args[1]["json"] == {
|
|
"model": "custom",
|
|
"params": {
|
|
"prompt": ["Hello, how are you?"],
|
|
"max_tokens": None,
|
|
"temperature": None,
|
|
"top_p": None,
|
|
"top_k": None,
|
|
},
|
|
"X-Custom-BodyValue": "custom-value",
|
|
"X-Custom-BodyValue2": "custom-value2",
|
|
}
|
|
|
|
# test that extra_body is not passed if not provided
|
|
with patch.object(
|
|
litellm.llms.custom_httpx.http_handler.HTTPHandler, "post"
|
|
) as mock_post:
|
|
response = litellm.completion(
|
|
model="custom/custom",
|
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
|
api_base="https://example.com/api/v1",
|
|
)
|
|
mock_post.assert_called_once()
|
|
assert mock_post.call_args[1]["json"] == {
|
|
"model": "custom",
|
|
"params": {
|
|
"prompt": ["Hello, how are you?"],
|
|
"max_tokens": None,
|
|
"temperature": None,
|
|
"top_p": None,
|
|
"top_k": None,
|
|
},
|
|
}
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def set_openrouter_api_key():
|
|
original_api_key = os.environ.get("OPENROUTER_API_KEY")
|
|
os.environ["OPENROUTER_API_KEY"] = "fake-key-for-testing"
|
|
yield
|
|
if original_api_key is not None:
|
|
os.environ["OPENROUTER_API_KEY"] = original_api_key
|
|
else:
|
|
del os.environ["OPENROUTER_API_KEY"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_extra_body_with_fallback(
|
|
respx_mock: respx.MockRouter, set_openrouter_api_key, monkeypatch
|
|
):
|
|
"""
|
|
test regression for https://github.com/BerriAI/litellm/issues/8425.
|
|
|
|
This was perhaps a wider issue with the acompletion function not passing kwargs such as extra_body correctly when fallbacks are specified.
|
|
"""
|
|
|
|
# Save original state to restore after test
|
|
original_disable_aiohttp = litellm.disable_aiohttp_transport
|
|
|
|
try:
|
|
# since this uses respx, we need to set use_aiohttp_transport to False
|
|
# Set both the global variable and environment variable to ensure it takes effect
|
|
litellm.disable_aiohttp_transport = True
|
|
monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True")
|
|
# Flush cache to ensure no stale aiohttp clients are used
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
# Set up test parameters
|
|
model = "openrouter/deepseek/deepseek-chat"
|
|
messages = [{"role": "user", "content": "Hello, world!"}]
|
|
extra_body = {
|
|
"provider": {
|
|
"order": ["DeepSeek"],
|
|
"allow_fallbacks": False,
|
|
"require_parameters": True,
|
|
}
|
|
}
|
|
fallbacks = [{"model": "openrouter/google/gemini-flash-1.5-8b"}]
|
|
|
|
# Set up mock to respond to any POST request to the OpenRouter endpoint
|
|
# This ensures it works for both primary and fallback models
|
|
mock_route = respx_mock.post("https://openrouter.ai/api/v1/chat/completions")
|
|
mock_route.return_value = httpx.Response(
|
|
200,
|
|
json={
|
|
"id": "chatcmpl-123",
|
|
"object": "chat.completion",
|
|
"created": 1677652288,
|
|
"model": model,
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "Hello from mocked response!",
|
|
},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
"usage": {
|
|
"prompt_tokens": 9,
|
|
"completion_tokens": 12,
|
|
"total_tokens": 21,
|
|
},
|
|
},
|
|
)
|
|
|
|
response = await litellm.acompletion(
|
|
model=model,
|
|
messages=messages,
|
|
extra_body=extra_body,
|
|
fallbacks=fallbacks,
|
|
api_key="fake-openrouter-api-key",
|
|
)
|
|
|
|
# Verify the response
|
|
assert response is not None
|
|
assert (
|
|
len(respx_mock.calls) > 0
|
|
), "Mock was not called - check if aiohttp transport is properly disabled"
|
|
|
|
# Get the request from the mock
|
|
request: httpx.Request = respx_mock.calls[0].request
|
|
request_body = request.read()
|
|
request_body = json.loads(request_body)
|
|
|
|
# Verify basic parameters
|
|
assert request_body["model"] == "deepseek/deepseek-chat"
|
|
assert request_body["messages"] == messages
|
|
|
|
# Verify the extra_body parameters remain under the provider key
|
|
assert request_body["provider"]["order"] == ["DeepSeek"]
|
|
assert request_body["provider"]["allow_fallbacks"] is False
|
|
assert request_body["provider"]["require_parameters"] is True
|
|
finally:
|
|
# Restore original state to prevent test pollution
|
|
litellm.disable_aiohttp_transport = original_disable_aiohttp
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
|
|
@pytest.mark.parametrize("env_base", ["OPENAI_BASE_URL", "OPENAI_API_BASE"])
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.flaky(retries=3, delay=1)
|
|
async def test_openai_env_base(
|
|
respx_mock: respx.MockRouter, env_base, openai_api_response, monkeypatch
|
|
):
|
|
"This tests OpenAI env variables are honored, including legacy OPENAI_API_BASE"
|
|
# Ensure aiohttp transport is disabled to use httpx which respx can mock
|
|
litellm.disable_aiohttp_transport = True
|
|
|
|
expected_base_url = "http://localhost:12345/v1"
|
|
|
|
# Assign the environment variable based on env_base, and use a fake API key.
|
|
monkeypatch.setenv(env_base, expected_base_url)
|
|
monkeypatch.setenv("OPENAI_API_KEY", "fake_openai_api_key")
|
|
|
|
model = "gpt-4o"
|
|
messages = [{"role": "user", "content": "Hello, how are you?"}]
|
|
|
|
# Configure respx mock to intercept the request
|
|
mock_route = respx_mock.post(
|
|
url__regex=r"http://localhost:12345/v1/chat/completions.*"
|
|
).mock(
|
|
return_value=httpx.Response(
|
|
status_code=200,
|
|
json={
|
|
"id": "chatcmpl-123",
|
|
"object": "chat.completion",
|
|
"created": 1677652288,
|
|
"model": model,
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "Hello from mocked response!",
|
|
},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
"usage": {
|
|
"prompt_tokens": 9,
|
|
"completion_tokens": 12,
|
|
"total_tokens": 21,
|
|
},
|
|
},
|
|
)
|
|
)
|
|
|
|
try:
|
|
response = await litellm.acompletion(model=model, messages=messages)
|
|
|
|
# verify we had a response
|
|
assert response.choices[0].message.content == "Hello from mocked response!"
|
|
|
|
# Verify the mock was called
|
|
assert (
|
|
mock_route.called
|
|
), "Mock route was not called - request may have bypassed respx"
|
|
finally:
|
|
# Clean up to avoid affecting other tests
|
|
litellm.disable_aiohttp_transport = False
|
|
|
|
|
|
def build_database_url(username, password, host, dbname):
|
|
username_enc = urllib.parse.quote_plus(username)
|
|
password_enc = urllib.parse.quote_plus(password)
|
|
dbname_enc = urllib.parse.quote_plus(dbname)
|
|
return f"postgresql://{username_enc}:{password_enc}@{host}/{dbname_enc}"
|
|
|
|
|
|
def test_build_database_url():
|
|
url = build_database_url("user@name", "p@ss:word", "localhost", "db/name")
|
|
assert url == "postgresql://user%40name:p%40ss%3Aword@localhost/db%2Fname"
|
|
|
|
|
|
def test_bedrock_llama():
|
|
litellm._turn_on_debug()
|
|
from litellm.types.utils import CallTypes
|
|
from litellm.utils import return_raw_request
|
|
|
|
model = "bedrock/invoke/us.meta.llama4-scout-17b-instruct-v1:0"
|
|
|
|
request = return_raw_request(
|
|
endpoint=CallTypes.completion,
|
|
kwargs={
|
|
"model": model,
|
|
"messages": [
|
|
{"role": "user", "content": "hi"},
|
|
],
|
|
},
|
|
)
|
|
print(request)
|
|
|
|
assert (
|
|
request["raw_request_body"]["prompt"]
|
|
== "<|begin_of_text|><|start_header_id|>user<|end_header_id|>\n\nhi<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n"
|
|
)
|
|
|
|
|
|
def _mocked_openai_chat_response(model: str) -> httpx.Response:
|
|
return httpx.Response(
|
|
status_code=200,
|
|
json={
|
|
"id": "chatcmpl-123",
|
|
"object": "chat.completion",
|
|
"created": 1677652288,
|
|
"model": model,
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "Hello from mocked response!",
|
|
},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
"usage": {
|
|
"prompt_tokens": 9,
|
|
"completion_tokens": 12,
|
|
"total_tokens": 21,
|
|
},
|
|
},
|
|
)
|
|
|
|
|
|
def test_completion_forwards_verbosity_in_raw_request(respx_mock: respx.MockRouter):
|
|
"""Regression test: completion() must forward the verbosity param to the provider request body."""
|
|
from litellm.types.utils import CallTypes
|
|
from litellm.utils import return_raw_request
|
|
|
|
model = "gpt-5.2"
|
|
messages = [{"role": "user", "content": "hi"}]
|
|
respx_mock.post("https://api.openai.com/v1/chat/completions").mock(
|
|
return_value=_mocked_openai_chat_response(model)
|
|
)
|
|
|
|
request = return_raw_request(
|
|
endpoint=CallTypes.completion,
|
|
kwargs={
|
|
"model": model,
|
|
"messages": messages,
|
|
"verbosity": "high",
|
|
},
|
|
)
|
|
|
|
assert request["raw_request_body"]["verbosity"] == "high"
|
|
assert request["raw_request_body"]["model"] == model
|
|
assert request["raw_request_body"]["messages"] == messages
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_acompletion_forwards_verbosity_to_provider_request(
|
|
respx_mock: respx.MockRouter, monkeypatch
|
|
):
|
|
"""Regression test: acompletion() must forward the verbosity param to the provider request body."""
|
|
original_disable_aiohttp = litellm.disable_aiohttp_transport
|
|
try:
|
|
litellm.disable_aiohttp_transport = True
|
|
monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True")
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
model = "gpt-5.2"
|
|
messages = [{"role": "user", "content": "hi"}]
|
|
mock_route = respx_mock.post("https://api.openai.com/v1/chat/completions").mock(
|
|
return_value=_mocked_openai_chat_response(model)
|
|
)
|
|
|
|
response = await litellm.acompletion(
|
|
model=model,
|
|
messages=messages,
|
|
verbosity="low",
|
|
api_key="fake-openai-api-key",
|
|
)
|
|
|
|
assert response.choices[0].message.content == "Hello from mocked response!"
|
|
assert mock_route.called
|
|
request_body = json.loads(respx_mock.calls[0].request.read())
|
|
assert request_body["verbosity"] == "low"
|
|
assert request_body["model"] == model
|
|
assert request_body["messages"] == messages
|
|
finally:
|
|
litellm.disable_aiohttp_transport = original_disable_aiohttp
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
|
|
def test_responses_api_bridge_check_strips_responses_prefix():
|
|
"""Test that responses_api_bridge_check strips 'responses/' prefix and sets mode."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 4096}
|
|
|
|
model_info, model = responses_api_bridge_check(
|
|
model="responses/gpt-4-responses",
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model == "gpt-4-responses"
|
|
assert model_info["mode"] == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_pro():
|
|
"""Test that gpt-5.4-pro routes through responses API bridge, not chat completions.
|
|
|
|
Regression test for https://github.com/BerriAI/litellm/issues/23014
|
|
gpt-5.4-pro is a responses-only model and must not be sent to /v1/chat/completions.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
for model_name in ["gpt-5.4-pro", "gpt-5.4-pro-2026-03-05"]:
|
|
model_info, model = responses_api_bridge_check(
|
|
model=model_name,
|
|
custom_llm_provider="openai",
|
|
)
|
|
assert (
|
|
model_info.get("mode") == "responses"
|
|
), f"{model_name} should have mode='responses', got '{model_info.get('mode')}'"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_tools_plus_reasoning_routes_to_responses():
|
|
"""gpt-5.4 with both tools and reasoning_effort should route to Responses API."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="xhigh",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_5_tools_plus_reasoning_routes_to_responses():
|
|
"""gpt-5.5+ with both tools and reasoning_effort should route to Responses API."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.5-pro",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="xhigh",
|
|
)
|
|
|
|
assert model == "gpt-5.5-pro"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_azure_gpt_5_4_tools_plus_reasoning_routes_to_responses():
|
|
"""Azure gpt-5.4 with both tools and reasoning_effort should route to Responses API."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="azure",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="high",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_azure_gpt_5_4_tools_with_default_reasoning_routes_to_responses():
|
|
"""
|
|
Azure gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables
|
|
reasoning by default for gpt-5.4+, and Chat Completions rejects function tools
|
|
whenever reasoning is on.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="azure",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_tools_with_default_reasoning_routes_to_responses():
|
|
"""
|
|
gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables reasoning
|
|
by default for gpt-5.4+, and Chat Completions rejects function tools whenever
|
|
reasoning is on ("use /v1/responses or set reasoning_effort to 'none'").
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
@pytest.mark.parametrize("model_name", ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra"])
|
|
def test_responses_api_bridge_check_gpt_5_6_tools_with_default_reasoning_routes_to_responses(
|
|
monkeypatch, model_name
|
|
):
|
|
"""
|
|
The whole gpt-5.6 family must bridge on function tools alone. The bridge used to
|
|
require an explicit reasoning_effort, so a gpt-5.6 call carrying tools and no effort
|
|
was rejected with "Function tools with reasoning_effort are not supported for
|
|
gpt-5.6-sol in /v1/chat/completions".
|
|
"""
|
|
import litellm
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
|
monkeypatch.setattr(litellm, "api_base", None)
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model=model_name,
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == model_name
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_tools_with_reasoning_none_stays_chat():
|
|
"""
|
|
Explicit reasoning_effort "none" is OpenAI's documented escape hatch that keeps
|
|
function tools servable on Chat Completions; the bridge must not fire.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="none",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_reasoning_none_with_summary_still_routes_to_responses():
|
|
"""A reasoning summary is Responses-only regardless of effort value."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
reasoning_effort="none",
|
|
reasoning_summary="detailed",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_custom_tools_only_stays_chat():
|
|
"""
|
|
Chat Completions serves custom (grammar) tools natively with reasoning on; only
|
|
FUNCTION tools trigger the OpenAI rejection. Custom-only requests must stay on chat
|
|
so responses keep the native custom tool_call shape instead of the bridge's
|
|
function-shaped mapping.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "custom", "custom": {"name": "ApplyPatch", "description": "V4A patch"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_mixed_function_and_custom_tools_routes_to_responses():
|
|
"""One function tool in the mix is enough to make chat unservable with reasoning on."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[
|
|
{"type": "custom", "custom": {"name": "ApplyPatch"}},
|
|
{"type": "function", "function": {"name": "shell"}},
|
|
],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_flat_function_tool_routes_to_responses():
|
|
"""Responses-style flat function tool defs still count as function tools."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "name": "shell", "parameters": {"type": "object"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_dict_effort_none_stays_chat():
|
|
"""The escape hatch must honor litellm's dict form: {"effort": "none"} means reasoning off."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort={"effort": "none"},
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_dict_effort_active_routes_to_responses():
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort={"effort": "low"},
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_dict_effort_none_with_summary_routes_to_responses():
|
|
"""A summary inside the dict form is Responses-only even when effort is none."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort={"effort": "none", "summary": "concise"},
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
@pytest.mark.parametrize("blank_api_base", [None, "", " ", "\t"])
|
|
def test_responses_api_bridge_check_blank_api_base_is_default_openai(blank_api_base):
|
|
"""
|
|
A blank api_base (None, empty, or whitespace) resolves to the default OpenAI
|
|
endpoint downstream, which enforces the reasoning+tools constraint, so gpt-5.4+
|
|
function-tool requests with unset reasoning_effort must still auto-bridge.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=blank_api_base,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_custom_api_base_with_unset_effort_stays_chat():
|
|
"""
|
|
Chat-only OpenAI-compatible backends registered under the openai provider with a
|
|
custom api_base and gpt-5.4+ model names serve tools-without-reasoning fine and
|
|
have no /responses route; the unset-effort arm must not reroute them.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base="http://vllm.internal:8000/v1",
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_custom_api_base_via_global_with_unset_effort_stays_chat(monkeypatch):
|
|
"""
|
|
A custom base set through the litellm.api_base global (not the call arg) is resolved the
|
|
same way the chat handler resolves it, so the unset-effort arm must not reroute a chat-only
|
|
backend to a /responses route it lacks. Regression guard: the gate previously inspected only
|
|
the call-level api_base and bridged these requests.
|
|
"""
|
|
import litellm
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
monkeypatch.setattr(litellm, "api_base", "http://vllm.internal:8000/v1")
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
@pytest.mark.parametrize("env_var", ["OPENAI_BASE_URL", "OPENAI_API_BASE"])
|
|
def test_responses_api_bridge_check_custom_api_base_via_env_with_unset_effort_stays_chat(monkeypatch, env_var):
|
|
"""
|
|
A custom base set via OPENAI_BASE_URL/OPENAI_API_BASE env is resolved identically to the chat
|
|
handler, so the unset-effort arm leaves the request on chat instead of bridging it.
|
|
"""
|
|
import litellm
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
monkeypatch.setattr(litellm, "api_base", None)
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
|
monkeypatch.setenv(env_var, "http://vllm.internal:8000/v1")
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_custom_api_base_with_explicit_effort_still_routes():
|
|
"""Explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="high",
|
|
api_base="http://vllm.internal:8000/v1",
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_azure_with_api_base_and_unset_effort_routes():
|
|
"""Azure OpenAI always sets api_base and does enforce the constraint; keep bridging."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="azure",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base="https://myresource.openai.azure.com",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat():
|
|
"""Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.1",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.1"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_reasoning_summary_without_tools_routes_to_responses():
|
|
"""gpt-5.4+ with reasoning_effort + reasoningSummary but no tools should bridge (AI SDK)."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
tools=None,
|
|
reasoning_effort="medium",
|
|
reasoning_summary="auto",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_reasoning_summary_routes_to_responses():
|
|
"""Bare ``gpt-5`` with reasoning_effort + reasoningSummary should bridge (not 5.4+)."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5",
|
|
custom_llm_provider="openai",
|
|
tools=None,
|
|
reasoning_effort="medium",
|
|
reasoning_summary="auto",
|
|
)
|
|
|
|
assert model == "gpt-5"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_tools_without_summary_stays_chat():
|
|
"""gpt-5 with tools + reasoning_effort but no summary should stay on chat."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="medium",
|
|
reasoning_summary=None,
|
|
)
|
|
|
|
assert model == "gpt-5"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
@patch("litellm.completion_extras.responses_api_bridge.completion")
|
|
def test_gpt_5_4_responses_bridge_preserves_reasoning_summary_dict(
|
|
mock_responses_completion,
|
|
):
|
|
"""When routed to Responses, preserve reasoning_effort summary dict."""
|
|
mock_responses_completion.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
litellm.completion(
|
|
model="gpt-5.4",
|
|
messages=[{"role": "user", "content": "What is the capital of France?"}],
|
|
tools=[
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "get_capital",
|
|
"description": "Get the capital of a country",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"country": {"type": "string"}},
|
|
},
|
|
},
|
|
}
|
|
],
|
|
reasoning_effort={"effort": "xhigh", "summary": "detailed"},
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert mock_responses_completion.called is True
|
|
optional_params = mock_responses_completion.call_args.kwargs["optional_params"]
|
|
assert optional_params["reasoning_effort"] == {
|
|
"effort": "xhigh",
|
|
"summary": "detailed",
|
|
}
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"model, model_info, expected_model_param, expected_base_model_param",
|
|
[
|
|
("gemini/gemini-3.1-pro", None, "gemini-3.1-pro", None),
|
|
(
|
|
"gemini/gemini-3.1-pro",
|
|
{"base_model": "gemini-3.1-pro-preview"},
|
|
"gemini-3.1-pro",
|
|
"gemini-3.1-pro-preview",
|
|
),
|
|
],
|
|
)
|
|
def test_completion_optional_params_base_model(
|
|
model: str,
|
|
model_info: dict | None,
|
|
expected_model_param: str,
|
|
expected_base_model_param: str | None,
|
|
):
|
|
"""``model_info.base_model`` must reach ``get_optional_params`` as ``base_model``
|
|
(an additive capability hint), without overwriting ``model`` with the label.
|
|
|
|
Regression for #29618: overwriting ``model`` with a friendly ``base_model``
|
|
label made Bedrock drop ``tools``/``tool_choice`` under ``drop_params``."""
|
|
with patch("litellm.main.get_optional_params") as mock_get_optional_params:
|
|
mock_get_optional_params.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
kwargs = {
|
|
"model": model,
|
|
"messages": [{"role": "user", "content": "What is the capital of France?"}],
|
|
"api_key": "fake-key",
|
|
"mock_response": "Hey, how's it going?",
|
|
}
|
|
if model_info is not None:
|
|
kwargs["model_info"] = model_info
|
|
|
|
litellm.completion(**kwargs)
|
|
|
|
assert mock_get_optional_params.called is True
|
|
call_kwargs = mock_get_optional_params.call_args.kwargs
|
|
assert call_kwargs["model"] == expected_model_param
|
|
assert call_kwargs["base_model"] == expected_base_model_param
|
|
|
|
|
|
@patch("litellm.completion_extras.responses_api_bridge.completion")
|
|
def test_gpt_5_4_responses_bridge_merges_reasoning_summary_kwarg_without_tools(
|
|
mock_responses_completion,
|
|
):
|
|
"""reasoningSummary without tools should route and merge into reasoning_effort dict."""
|
|
mock_responses_completion.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
litellm.completion(
|
|
model="gpt-5.4",
|
|
messages=[{"role": "user", "content": "ok"}],
|
|
reasoning_effort="medium",
|
|
reasoningSummary="auto",
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert mock_responses_completion.called is True
|
|
optional_params = mock_responses_completion.call_args.kwargs["optional_params"]
|
|
assert optional_params["reasoning_effort"] == {
|
|
"effort": "medium",
|
|
"summary": "auto",
|
|
}
|
|
assert "reasoningSummary" not in optional_params
|
|
assert "reasoning_summary" not in optional_params
|
|
|
|
|
|
@patch("litellm.completion_extras.responses_api_bridge.completion")
|
|
def test_responses_bridge_preserves_reasoning_summary_without_effort(
|
|
mock_responses_completion,
|
|
):
|
|
"""Reasoning summary should survive responses routing even without effort."""
|
|
mock_responses_completion.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
with patch.object(litellm, "route_all_chat_openai_to_responses", True):
|
|
litellm.completion(
|
|
model="gpt-4o",
|
|
messages=[{"role": "user", "content": "ok"}],
|
|
reasoningSummary="auto",
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert mock_responses_completion.called is True
|
|
optional_params = mock_responses_completion.call_args.kwargs["optional_params"]
|
|
assert optional_params["reasoning_effort"] == {"summary": "auto"}
|
|
assert "reasoningSummary" not in optional_params
|
|
assert "reasoning_summary" not in optional_params
|
|
|
|
|
|
@patch("litellm.completion_extras.responses_api_bridge.completion")
|
|
def test_gpt_5_responses_bridge_tools_and_reasoning_summary(
|
|
mock_responses_completion,
|
|
):
|
|
"""Bare gpt-5 with tools + reasoningSummary should bridge (OpenCode-style)."""
|
|
mock_responses_completion.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
litellm.completion(
|
|
model="gpt-5",
|
|
messages=[{"role": "user", "content": "ok"}],
|
|
tools=[
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "apply_patch",
|
|
"parameters": {"type": "object", "properties": {}},
|
|
},
|
|
}
|
|
],
|
|
tool_choice="auto",
|
|
reasoning_effort="medium",
|
|
reasoningSummary="auto",
|
|
stream=True,
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert mock_responses_completion.called is True
|
|
optional_params = mock_responses_completion.call_args.kwargs["optional_params"]
|
|
assert optional_params.get("reasoning_effort") == {
|
|
"effort": "medium",
|
|
"summary": "auto",
|
|
}
|
|
|
|
|
|
def test_responses_api_bridge_check_handles_exception():
|
|
"""Test that responses_api_bridge_check handles exceptions and still processes responses/ models."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.side_effect = Exception("Model not found")
|
|
|
|
model_info, model = responses_api_bridge_check(
|
|
model="responses/custom-model", custom_llm_provider="custom"
|
|
)
|
|
|
|
assert model == "custom-model"
|
|
assert model_info["mode"] == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_global_flag_routes_openai():
|
|
"""When route_all_chat_openai_to_responses is True, any OpenAI model routes to responses."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch.object(litellm, "route_all_chat_openai_to_responses", True):
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-4o",
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model == "gpt-4o"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_global_flag_does_not_affect_azure():
|
|
"""route_all_chat_openai_to_responses should not affect Azure models."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch.object(litellm, "route_all_chat_openai_to_responses", True):
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 4096}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-4o",
|
|
custom_llm_provider="azure",
|
|
)
|
|
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_global_flag_default_false():
|
|
"""By default, route_all_chat_openai_to_responses is False and doesn't affect routing."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch.object(litellm, "route_all_chat_openai_to_responses", False):
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 4096}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-4o",
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_async_mock_delay():
|
|
"""Use asyncio await for mock delay on acompletion"""
|
|
import time
|
|
|
|
from litellm import acompletion
|
|
|
|
start_time = time.time()
|
|
result = await acompletion(
|
|
model="gpt-3.5-turbo",
|
|
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
|
mock_delay=0.01,
|
|
mock_response="Hello world",
|
|
)
|
|
end_time = time.time()
|
|
delay = end_time - start_time
|
|
assert delay >= 0.01
|
|
|
|
|
|
def test_stream_chunk_builder_thinking_blocks():
|
|
from litellm import stream_chunk_builder
|
|
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
|
|
|
|
chunks = [
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content="I need to summar",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "I need to summar",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "I need to summar",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role="assistant",
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content="ize the previous agent's thinking process into a",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "ize the previous agent's thinking process into a",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "ize the previous agent's thinking process into a",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content=" short description. Based on the input data provide",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " short description. Based on the input data provide",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " short description. Based on the input data provide",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content="d, it seems the agent was planning to refine their search",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "d, it seems the agent was planning to refine their search",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "d, it seems the agent was planning to refine their search",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content=" to focus more on technical aspects of home automation and home",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " to focus more on technical aspects of home automation and home",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " to focus more on technical aspects of home automation and home",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content=" energy system management.\n\nI'll create a brief",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " energy system management.\n\nI'll create a brief",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " energy system management.\n\nI'll create a brief",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content=" summary of what the agent was doing.",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " summary of what the agent was doing.",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " summary of what the agent was doing.",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content="",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "",
|
|
"signature": "ErUBCkYIBRgCIkAKBSMkB2+MBF643wiWxlERsGXVdlhbPx9lnTIbygzjFIeZ5uhTV+HNWDon9vQV4hmXvAKwQfwS8vkNFB366l05Egzt2U18IpRrZRyQn1UaDDdYvKHYP8Ps1IbWjSIw8eSYOU9gtqNcwR6D0wY7iOPx2GliDEatLI5rSs96CByoTIoADL2M5bX8KP0jEpbHKh0ccYryigdH/3J8EiFt/BmGUceVASP5l9r22dFWiBgC",
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "",
|
|
"signature": "ErUBCkYIBRgCIkAKBSMkB2+MBF643wiWxlERsGXVdlhbPx9lnTIbygzjFIeZ5uhTV+HNWDon9vQV4hmXvAKwQfwS8vkNFB366l05Egzt2U18IpRrZRyQn1UaDDdYvKHYP8Ps1IbWjSIw8eSYOU9gtqNcwR6D0wY7iOPx2GliDEatLI5rSs96CByoTIoADL2M5bX8KP0jEpbHKh0ccYryigdH/3J8EiFt/BmGUceVASP5l9r22dFWiBgC",
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content='{"a',
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content='gent_doing"',
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content=': "Re',
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content="searching",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content=" technic",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content="al aspect",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content="s of home au",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content='tomation"}',
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason="tool_calls",
|
|
index=0,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content=None,
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
),
|
|
]
|
|
|
|
response = stream_chunk_builder(chunks=chunks)
|
|
print(response)
|
|
|
|
assert response is not None
|
|
assert response.choices[0].message.content is not None
|
|
assert response.choices[0].message.thinking_blocks is not None
|
|
|
|
|
|
from litellm.llms.openai.openai import OpenAIChatCompletion
|
|
|
|
|
|
def throw_retryable_error(*_, **__):
|
|
raise RuntimeError("BOOM")
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_retrying() -> None:
|
|
litellm.num_retries = 10
|
|
with (
|
|
patch.object(
|
|
OpenAIChatCompletion,
|
|
"make_openai_chat_completion_request",
|
|
side_effect=throw_retryable_error,
|
|
) as mock_request,
|
|
pytest.raises(litellm.InternalServerError, match="LiteLLM Retried: 10 times"),
|
|
):
|
|
await litellm.acompletion(
|
|
model="gpt-4o-mini",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
)
|
|
|
|
|
|
def test_anthropic_disable_url_suffix_env_var():
|
|
"""Test that LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX prevents /v1/messages suffix."""
|
|
import os
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
from litellm import completion
|
|
|
|
# Test with environment variable disabled (default behavior)
|
|
with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}):
|
|
actual_api_base = None
|
|
|
|
with patch("litellm.main.anthropic_chat_completions") as mock_anthropic:
|
|
|
|
def capture_completion(**kwargs):
|
|
nonlocal actual_api_base
|
|
actual_api_base = kwargs.get("api_base")
|
|
mock_response = MagicMock()
|
|
mock_response.choices = [MagicMock()]
|
|
return mock_response
|
|
|
|
mock_anthropic.completion = capture_completion
|
|
|
|
# This should append /v1/messages
|
|
completion(
|
|
model="anthropic/claude-3-sonnet",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
api_key="test-key",
|
|
)
|
|
|
|
# Verify the api_base has /v1/messages appended
|
|
assert actual_api_base.endswith("/v1/messages")
|
|
assert actual_api_base == "https://api.example.com/v1/messages"
|
|
|
|
# Test with environment variable enabled
|
|
with patch.dict(
|
|
os.environ,
|
|
{
|
|
"ANTHROPIC_API_BASE": "https://api.example.com/custom/path",
|
|
"LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX": "true",
|
|
},
|
|
):
|
|
actual_api_base = None
|
|
|
|
with patch("litellm.main.anthropic_chat_completions") as mock_anthropic:
|
|
|
|
def capture_completion(**kwargs):
|
|
nonlocal actual_api_base
|
|
actual_api_base = kwargs.get("api_base")
|
|
mock_response = MagicMock()
|
|
mock_response.choices = [MagicMock()]
|
|
return mock_response
|
|
|
|
mock_anthropic.completion = capture_completion
|
|
|
|
# This should NOT append /v1/messages
|
|
completion(
|
|
model="anthropic/claude-3-sonnet",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
api_key="test-key",
|
|
)
|
|
|
|
# Verify the api_base does not have /v1/messages appended
|
|
assert actual_api_base == "https://api.example.com/custom/path"
|
|
assert not actual_api_base.endswith("/v1/messages")
|
|
|
|
|
|
def test_anthropic_text_disable_url_suffix_env_var():
|
|
"""Test that LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX prevents /v1/complete suffix for anthropic_text."""
|
|
import os
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
from litellm import completion
|
|
|
|
# Test with environment variable disabled (default behavior)
|
|
with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}):
|
|
actual_api_base = None
|
|
|
|
with patch("litellm.main.base_llm_http_handler") as mock_handler:
|
|
|
|
def capture_completion(**kwargs):
|
|
nonlocal actual_api_base
|
|
actual_api_base = kwargs.get("api_base")
|
|
return MagicMock()
|
|
|
|
mock_handler.completion = capture_completion
|
|
|
|
# This should append /v1/complete
|
|
completion(
|
|
model="anthropic_text/claude-instant-1",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
api_key="test-key",
|
|
)
|
|
|
|
# Verify the api_base has /v1/complete appended
|
|
assert actual_api_base.endswith("/v1/complete")
|
|
assert actual_api_base == "https://api.example.com/v1/complete"
|
|
|
|
# Test with environment variable enabled
|
|
with patch.dict(
|
|
os.environ,
|
|
{
|
|
"ANTHROPIC_API_BASE": "https://api.example.com/custom/complete",
|
|
"LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX": "true",
|
|
},
|
|
):
|
|
actual_api_base = None
|
|
|
|
with patch("litellm.main.base_llm_http_handler") as mock_handler:
|
|
|
|
def capture_completion(**kwargs):
|
|
nonlocal actual_api_base
|
|
actual_api_base = kwargs.get("api_base")
|
|
return MagicMock()
|
|
|
|
mock_handler.completion = capture_completion
|
|
|
|
# This should NOT append /v1/complete
|
|
completion(
|
|
model="anthropic_text/claude-instant-1",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
api_key="test-key",
|
|
)
|
|
|
|
# Verify the api_base does not have /v1/complete appended
|
|
assert actual_api_base == "https://api.example.com/custom/complete"
|
|
assert not actual_api_base.endswith("/v1/complete")
|
|
|
|
|
|
def test_image_edit_merges_headers_and_extra_headers():
|
|
from litellm.images.main import base_llm_http_handler
|
|
|
|
combined_headers = {
|
|
"x-test-header-one": "value-1",
|
|
"x-test-header-two": "value-2",
|
|
}
|
|
|
|
mock_image_edit_config = MagicMock()
|
|
mock_image_edit_config.get_supported_openai_params.return_value = set()
|
|
mock_image_edit_config.map_openai_params.side_effect = lambda **kwargs: dict(
|
|
kwargs["image_edit_optional_params"]
|
|
)
|
|
|
|
with (
|
|
patch(
|
|
"litellm.images.main.ProviderConfigManager.get_provider_image_edit_config",
|
|
return_value=mock_image_edit_config,
|
|
) as mock_config,
|
|
patch.object(
|
|
base_llm_http_handler,
|
|
"image_edit_handler",
|
|
return_value="ok",
|
|
) as mock_handler,
|
|
):
|
|
response = litellm.image_edit(
|
|
image=MagicMock(name="image"),
|
|
prompt="test",
|
|
model="azure/gpt-image-1",
|
|
headers={"x-test-header-one": "value-1"},
|
|
extra_headers={
|
|
"x-test-header-two": "value-2",
|
|
},
|
|
)
|
|
|
|
assert response == "ok"
|
|
mock_config.assert_called_once()
|
|
|
|
handler_kwargs = mock_handler.call_args.kwargs
|
|
assert handler_kwargs["extra_headers"] == combined_headers
|
|
assert "extra_headers" not in handler_kwargs["image_edit_optional_request_params"]
|
|
|
|
|
|
def test_mock_completion_stream_with_model_response():
|
|
"""Test that mock_completion correctly handles stream=True with a ModelResponse as mock_response."""
|
|
from litellm import completion
|
|
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
|
|
|
# Create a ModelResponse object
|
|
mock_model_response = ModelResponse(
|
|
id="chatcmpl-test-123",
|
|
created=1234567890,
|
|
model="gpt-4o-mini",
|
|
object="chat.completion",
|
|
choices=[
|
|
Choices(
|
|
finish_reason="stop",
|
|
index=0,
|
|
message=Message(
|
|
content="This is a test response",
|
|
role="assistant",
|
|
),
|
|
)
|
|
],
|
|
usage=Usage(
|
|
prompt_tokens=10,
|
|
completion_tokens=20,
|
|
total_tokens=30,
|
|
),
|
|
)
|
|
|
|
# Call completion with stream=True and mock_response as ModelResponse
|
|
response = completion(
|
|
model="gpt-4o-mini",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
stream=True,
|
|
mock_response=mock_model_response,
|
|
)
|
|
|
|
# Verify that the response is a stream
|
|
assert response is not None
|
|
|
|
# Collect all chunks from the stream
|
|
chunks = []
|
|
for chunk in response:
|
|
chunks.append(chunk)
|
|
print(f"Chunk: {chunk}")
|
|
|
|
# Verify we got chunks
|
|
assert len(chunks) > 0
|
|
|
|
# Verify the content is streamed correctly
|
|
accumulated_content = ""
|
|
for chunk in chunks:
|
|
if (
|
|
hasattr(chunk.choices[0].delta, "content")
|
|
and chunk.choices[0].delta.content
|
|
):
|
|
accumulated_content += chunk.choices[0].delta.content
|
|
|
|
assert "This is a test response" in accumulated_content or len(chunks) > 0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_async_mock_completion_stream_with_model_response():
|
|
"""Test that async mock_completion correctly handles stream=True with a ModelResponse as mock_response."""
|
|
from litellm import acompletion
|
|
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
|
|
|
# Create a ModelResponse object
|
|
mock_model_response = ModelResponse(
|
|
id="chatcmpl-test-456",
|
|
created=1234567890,
|
|
model="gpt-4o-mini",
|
|
object="chat.completion",
|
|
choices=[
|
|
Choices(
|
|
finish_reason="stop",
|
|
index=0,
|
|
message=Message(
|
|
content="This is an async test response",
|
|
role="assistant",
|
|
),
|
|
)
|
|
],
|
|
usage=Usage(
|
|
prompt_tokens=15,
|
|
completion_tokens=25,
|
|
total_tokens=40,
|
|
),
|
|
)
|
|
|
|
# Call acompletion with stream=True and mock_response as ModelResponse
|
|
response = await acompletion(
|
|
model="gpt-4o-mini",
|
|
messages=[{"role": "user", "content": "Hello async"}],
|
|
stream=True,
|
|
mock_response=mock_model_response,
|
|
)
|
|
|
|
# Verify that the response is a stream
|
|
assert response is not None
|
|
|
|
# Collect all chunks from the stream
|
|
chunks = []
|
|
async for chunk in response:
|
|
chunks.append(chunk)
|
|
print(f"Async Chunk: {chunk}")
|
|
|
|
# Verify we got chunks
|
|
assert len(chunks) > 0
|
|
|
|
# Verify the content is streamed correctly
|
|
accumulated_content = ""
|
|
for chunk in chunks:
|
|
if (
|
|
hasattr(chunk.choices[0].delta, "content")
|
|
and chunk.choices[0].delta.content
|
|
):
|
|
accumulated_content += chunk.choices[0].delta.content
|
|
|
|
assert "This is an async test response" in accumulated_content or len(chunks) > 0
|
|
|
|
|
|
class TestCallTypesOCR:
|
|
"""Test that OCR call types are properly defined in CallTypes enum.
|
|
|
|
Fixes https://github.com/BerriAI/litellm/issues/17381
|
|
"""
|
|
|
|
def test_ocr_call_type_exists(self):
|
|
"""Test that CallTypes.ocr exists and has correct value."""
|
|
from litellm.types.utils import CallTypes
|
|
|
|
assert hasattr(CallTypes, "ocr")
|
|
assert CallTypes.ocr.value == "ocr"
|
|
|
|
def test_aocr_call_type_exists(self):
|
|
"""Test that CallTypes.aocr exists and has correct value."""
|
|
from litellm.types.utils import CallTypes
|
|
|
|
assert hasattr(CallTypes, "aocr")
|
|
assert CallTypes.aocr.value == "aocr"
|
|
|
|
def test_ocr_call_type_from_string(self):
|
|
"""Test that CallTypes can be constructed from 'ocr' string."""
|
|
from litellm.types.utils import CallTypes
|
|
|
|
call_type = CallTypes("ocr")
|
|
assert call_type == CallTypes.ocr
|
|
|
|
def test_aocr_call_type_from_string(self):
|
|
"""Test that CallTypes can be constructed from 'aocr' string.
|
|
|
|
This is the actual use case that was failing - the OCR endpoint
|
|
uses route_type='aocr' and guardrails try to instantiate
|
|
CallTypes('aocr').
|
|
"""
|
|
from litellm.types.utils import CallTypes
|
|
|
|
call_type = CallTypes("aocr")
|
|
assert call_type == CallTypes.aocr
|
|
|
|
|
|
def test_stream_chunk_builder_text_completion_combines_text_and_usage():
|
|
from litellm.main import stream_chunk_builder_text_completion
|
|
from litellm.types.utils import TextCompletionResponse
|
|
|
|
chunks = [
|
|
TextCompletionResponse(
|
|
id="cmpl-1",
|
|
object="text_completion",
|
|
created=1,
|
|
model="gpt-3.5-turbo-instruct",
|
|
choices=[{"text": "Hello", "index": 0, "logprobs": None, "finish_reason": None}],
|
|
),
|
|
TextCompletionResponse(
|
|
id="cmpl-1",
|
|
object="text_completion",
|
|
created=1,
|
|
model="gpt-3.5-turbo-instruct",
|
|
choices=[{"text": " world", "index": 0, "logprobs": None, "finish_reason": "stop"}],
|
|
),
|
|
]
|
|
|
|
response = stream_chunk_builder_text_completion(
|
|
chunks=chunks, messages=[{"role": "user", "content": "say hello"}]
|
|
)
|
|
|
|
assert response.choices[0].text == "Hello world"
|
|
assert response.choices[0].finish_reason == "stop"
|
|
assert response.usage.prompt_tokens > 0
|
|
assert response.usage.completion_tokens > 0
|
|
assert response.usage.total_tokens == response.usage.prompt_tokens + response.usage.completion_tokens
|
|
|
|
|
|
def test_completion_forwards_store_and_prompt_cache_key_to_openai():
|
|
"""
|
|
Regression test for https://github.com/BerriAI/litellm/issues/33184
|
|
|
|
store and prompt_cache_key are documented OpenAI chat completion params that
|
|
were accepted as supported but silently dropped before the provider request
|
|
was built, because they were not named parameters of completion() and
|
|
get_optional_params() the way safety_identifier is.
|
|
"""
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key")
|
|
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_client:
|
|
try:
|
|
litellm.completion(
|
|
model="openai/gpt-4o",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
store=False,
|
|
prompt_cache_key="test-cache-key",
|
|
client=client,
|
|
)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_client.assert_called_once()
|
|
request_body = mock_client.call_args.kwargs
|
|
assert request_body["store"] is False
|
|
assert request_body["prompt_cache_key"] == "test-cache-key"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_acompletion_forwards_store_and_prompt_cache_key_to_openai():
|
|
"""
|
|
Async variant of the store/prompt_cache_key forwarding regression test for
|
|
https://github.com/BerriAI/litellm/issues/33184
|
|
"""
|
|
from openai import AsyncOpenAI
|
|
|
|
client = AsyncOpenAI(api_key="fake-api-key")
|
|
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_client:
|
|
try:
|
|
await litellm.acompletion(
|
|
model="openai/gpt-4o",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
store=False,
|
|
prompt_cache_key="test-cache-key",
|
|
client=client,
|
|
)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_client.assert_called_once()
|
|
request_body = mock_client.call_args.kwargs
|
|
assert request_body["store"] is False
|
|
assert request_body["prompt_cache_key"] == "test-cache-key"
|
|
|
|
|
|
def test_completion_omits_store_and_prompt_cache_key_when_not_passed():
|
|
"""
|
|
When store and prompt_cache_key are not passed, they must not appear in the
|
|
outbound request body (guards against always forwarding None defaults).
|
|
"""
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key")
|
|
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_client:
|
|
try:
|
|
litellm.completion(
|
|
model="openai/gpt-4o",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
client=client,
|
|
)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_client.assert_called_once()
|
|
request_body = mock_client.call_args.kwargs
|
|
assert "store" not in request_body
|
|
assert "prompt_cache_key" not in request_body
|
|
|
|
|
|
def test_completion_forwards_store_and_prompt_cache_key_to_mcp_gateway():
|
|
"""
|
|
Regression test for the MCP gateway early-return in completion(): store and
|
|
prompt_cache_key are named params, so they no longer travel via **kwargs and
|
|
must be forwarded explicitly like safety_identifier and service_tier.
|
|
"""
|
|
with patch(
|
|
"litellm.responses.mcp.chat_completions_handler.acompletion_with_mcp"
|
|
) as mock_mcp:
|
|
result = litellm.completion(
|
|
model="openai/gpt-4o",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
tools=[{"type": "mcp", "server_url": "litellm_proxy"}],
|
|
store=False,
|
|
prompt_cache_key="test-cache-key",
|
|
)
|
|
|
|
result.close()
|
|
mock_mcp.assert_called_once()
|
|
call_kwargs = mock_mcp.call_args.kwargs
|
|
assert call_kwargs["store"] is False
|
|
assert call_kwargs["prompt_cache_key"] == "test-cache-key"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(
|
|
"aws_credential_kwargs",
|
|
[
|
|
{
|
|
"aws_session_name": "litellm-gcp",
|
|
"aws_role_name": "arn:aws:iam::123456789012:role/litellm-bedrock-role",
|
|
"aws_web_identity_token": "oidc/google/108963886734710037768",
|
|
},
|
|
{
|
|
"aws_access_key_id": "AKIASTATICKEYFORTEST",
|
|
"aws_secret_access_key": "static-secret-key",
|
|
"aws_session_token": "static-session-token",
|
|
},
|
|
],
|
|
ids=["web_identity", "static_keys"],
|
|
)
|
|
async def test_acompletion_forwards_aws_credentials_through_responses_bridge(
|
|
respx_mock: respx.MockRouter, monkeypatch, aws_credential_kwargs: dict
|
|
):
|
|
from botocore.credentials import Credentials
|
|
|
|
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
|
|
|
|
original_disable_aiohttp = litellm.disable_aiohttp_transport
|
|
try:
|
|
litellm.disable_aiohttp_transport = True
|
|
monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True")
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False)
|
|
monkeypatch.delenv("BEDROCK_MANTLE_API_KEY", raising=False)
|
|
|
|
get_credentials_mock = MagicMock(return_value=Credentials("fake-key", "fake-secret"))
|
|
monkeypatch.setattr(BaseAWSLLM, "get_credentials", get_credentials_mock)
|
|
|
|
respx_mock.post("https://bedrock-mantle.us-east-2.api.aws/openai/v1/responses").respond(
|
|
json={
|
|
"id": "resp_123",
|
|
"object": "response",
|
|
"created_at": 1760144904,
|
|
"status": "completed",
|
|
"model": "openai.gpt-5.4",
|
|
"output": [
|
|
{
|
|
"type": "message",
|
|
"id": "msg_1",
|
|
"role": "assistant",
|
|
"status": "completed",
|
|
"content": [{"type": "output_text", "text": "ok", "annotations": []}],
|
|
}
|
|
],
|
|
}
|
|
)
|
|
|
|
response = await litellm.acompletion(
|
|
model="bedrock_mantle/openai.gpt-5.4",
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
api_base="https://bedrock-mantle.us-east-2.api.aws/v1",
|
|
aws_region_name="us-east-2",
|
|
num_retries=0,
|
|
**aws_credential_kwargs,
|
|
)
|
|
|
|
assert response.choices[0].message.content == "ok"
|
|
credential_kwargs = get_credentials_mock.call_args.kwargs
|
|
assert credential_kwargs["aws_region_name"] == "us-east-2"
|
|
for key, value in aws_credential_kwargs.items():
|
|
assert credential_kwargs[key] == value
|
|
authorization = respx_mock.calls.last.request.headers["Authorization"]
|
|
assert authorization.startswith("AWS4-HMAC-SHA256")
|
|
assert "fake-key" in authorization
|
|
finally:
|
|
litellm.disable_aiohttp_transport = original_disable_aiohttp
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
|
|
_GEMINI_RESPONSE_BODY = {
|
|
"candidates": [{"content": {"parts": [{"text": "hello"}], "role": "model"}, "finishReason": "STOP"}],
|
|
"usageMetadata": {"promptTokenCount": 2, "candidatesTokenCount": 1, "totalTokenCount": 3},
|
|
}
|
|
|
|
|
|
def _gemini_client_returning_a_reply():
|
|
"""An injected HTTP client whose post() answers like generativelanguage does."""
|
|
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
|
|
|
client = HTTPHandler()
|
|
request = httpx.Request("POST", "https://generativelanguage.googleapis.com/")
|
|
post = MagicMock(return_value=httpx.Response(200, json=_GEMINI_RESPONSE_BODY, request=request))
|
|
return client, post
|
|
|
|
|
|
@pytest.fixture
|
|
def restore_model_registry():
|
|
"""litellm.model_cost and the provider name sets are module-global.
|
|
|
|
register_model merges into the existing entry in place, hence the deep copy.
|
|
"""
|
|
model_cost = copy.deepcopy(litellm.model_cost)
|
|
openai_models = set(litellm.open_ai_chat_completion_models)
|
|
yield
|
|
litellm.model_cost.clear()
|
|
litellm.model_cost.update(model_cost)
|
|
litellm.open_ai_chat_completion_models.clear()
|
|
litellm.open_ai_chat_completion_models.update(openai_models)
|
|
|
|
|
|
def test_openai_model_name_does_not_outrank_explicit_provider():
|
|
"""`gemini/gpt-4o` goes to Google, not to litellm's OpenAI handler.
|
|
|
|
completion() checks `model in litellm.open_ai_chat_completion_models` ahead of
|
|
the gemini branch, so the call used to reach the OpenAI handler carrying
|
|
VertexGeminiConfig, whose transform_request raises NotImplementedError.
|
|
"""
|
|
assert "gpt-4o" in litellm.open_ai_chat_completion_models
|
|
client, post = _gemini_client_returning_a_reply()
|
|
|
|
with patch.object(client, "post", new=post):
|
|
response = litellm.completion(
|
|
model="gemini/gpt-4o",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
api_key="test-api-key",
|
|
client=client,
|
|
)
|
|
|
|
assert "generativelanguage.googleapis.com" in post.call_args.kwargs["url"]
|
|
assert "models/gpt-4o" in post.call_args.kwargs["url"]
|
|
assert response.choices[0].message.content == "hello"
|
|
|
|
|
|
def test_mislabelled_pricing_entry_does_not_reroute_provider(restore_model_registry):
|
|
"""register_model is the other way into the same failure.
|
|
|
|
An entry claiming litellm_provider "openai" adds its name to
|
|
open_ai_chat_completion_models, so one mislabelled price reroutes every later
|
|
call to that model in the process.
|
|
"""
|
|
litellm.register_model(
|
|
{
|
|
"gemini-2.5-pro": {
|
|
"litellm_provider": "openai",
|
|
"mode": "chat",
|
|
"input_cost_per_token": 1e-06,
|
|
"output_cost_per_token": 4e-06,
|
|
}
|
|
}
|
|
)
|
|
assert "gemini-2.5-pro" in litellm.open_ai_chat_completion_models
|
|
client, post = _gemini_client_returning_a_reply()
|
|
|
|
with patch.object(client, "post", new=post):
|
|
response = litellm.completion(
|
|
model="gemini/gemini-2.5-pro",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
api_key="test-api-key",
|
|
client=client,
|
|
)
|
|
|
|
assert "generativelanguage.googleapis.com" in post.call_args.kwargs["url"]
|
|
assert response.choices[0].message.content == "hello"
|
|
|
|
|
|
def test_openai_model_without_a_provider_still_routes_to_openai():
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-key")
|
|
raw_response = client.chat.completions.with_raw_response
|
|
with patch.object(raw_response, "create") as mock_create, contextlib.suppress(Exception):
|
|
litellm.completion(
|
|
model="gpt-4o",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
client=client,
|
|
)
|
|
|
|
mock_create.assert_called()
|
|
|
|
|
|
def _openai_chat_create_kwargs(client, **completion_kwargs):
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_client:
|
|
with contextlib.suppress(Exception):
|
|
litellm.completion(
|
|
messages=[{"role": "system", "content": "sys"}, {"role": "user", "content": "hi"}],
|
|
cache_control_injection_points=[{"location": "message", "role": "system"}],
|
|
client=client,
|
|
**completion_kwargs,
|
|
)
|
|
|
|
mock_client.assert_called_once()
|
|
return mock_client.call_args.kwargs
|
|
|
|
|
|
@pytest.fixture
|
|
def _no_openai_api_base_override(monkeypatch):
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
|
monkeypatch.setattr(litellm, "api_base", None)
|
|
|
|
|
|
@pytest.mark.usefixtures("_no_openai_api_base_override")
|
|
def test_completion_custom_api_base_sends_no_prompt_cache_breakpoint_for_gpt_5_6():
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key", base_url="http://127.0.0.1:9/v1")
|
|
request_body = _openai_chat_create_kwargs(client, model="gpt-5.6", api_base="http://127.0.0.1:9/v1")
|
|
|
|
assert request_body["messages"][0] == {"role": "system", "content": "sys", "cache_control": {"type": "ephemeral"}}
|
|
assert "prompt_cache_breakpoint" not in json.dumps(request_body["messages"])
|
|
assert "prompt_cache_options" not in json.dumps(request_body)
|
|
|
|
|
|
@pytest.mark.usefixtures("_no_openai_api_base_override")
|
|
def test_completion_custom_base_url_sends_no_prompt_cache_breakpoint_for_gpt_5_6():
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key", base_url="http://127.0.0.1:9/v1")
|
|
request_body = _openai_chat_create_kwargs(client, model="gpt-5.6", base_url="http://127.0.0.1:9/v1")
|
|
|
|
assert request_body["messages"][0] == {"role": "system", "content": "sys", "cache_control": {"type": "ephemeral"}}
|
|
assert "prompt_cache_breakpoint" not in json.dumps(request_body["messages"])
|
|
assert "prompt_cache_options" not in json.dumps(request_body)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.usefixtures("_no_openai_api_base_override")
|
|
async def test_acompletion_custom_base_url_sends_no_prompt_cache_breakpoint_for_gpt_5_6():
|
|
from openai import AsyncOpenAI
|
|
|
|
client = AsyncOpenAI(api_key="fake-api-key", base_url="http://127.0.0.1:9/v1")
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_create:
|
|
with contextlib.suppress(Exception):
|
|
await litellm.acompletion(
|
|
model="gpt-5.6",
|
|
messages=[{"role": "system", "content": "sys"}, {"role": "user", "content": "hi"}],
|
|
cache_control_injection_points=[{"location": "message", "role": "system"}],
|
|
client=client,
|
|
base_url="http://127.0.0.1:9/v1",
|
|
)
|
|
|
|
mock_create.assert_called_once()
|
|
request_body = mock_create.call_args.kwargs
|
|
|
|
assert request_body["messages"][0] == {"role": "system", "content": "sys", "cache_control": {"type": "ephemeral"}}
|
|
assert "prompt_cache_breakpoint" not in json.dumps(request_body["messages"])
|
|
assert "prompt_cache_options" not in json.dumps(request_body)
|
|
|
|
|
|
@pytest.mark.usefixtures("_no_openai_api_base_override")
|
|
def test_completion_default_api_base_sends_prompt_cache_breakpoint_for_gpt_5_6():
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key")
|
|
request_body = _openai_chat_create_kwargs(client, model="gpt-5.6")
|
|
|
|
assert request_body["messages"][0]["content"] == [
|
|
{"type": "text", "text": "sys", "prompt_cache_breakpoint": {"mode": "explicit"}}
|
|
]
|
|
assert request_body["extra_body"]["prompt_cache_options"] == {"mode": "explicit"}
|
|
|
|
|
|
_SUBSCRIPTION_OAUTH_CREDENTIAL = "Bearer sk-ant-oat01-fake-subscription-token-for-testing-0123456789"
|
|
|
|
|
|
def _scoped_headers_for_oauth_request():
|
|
from litellm.types.utils import ProviderSpecificHeader
|
|
|
|
return [
|
|
ProviderSpecificHeader(
|
|
custom_llm_provider="anthropic,bedrock,vertex_ai",
|
|
extra_headers={"anthropic-version": "2023-06-01"},
|
|
),
|
|
ProviderSpecificHeader(
|
|
custom_llm_provider="anthropic",
|
|
extra_headers={"authorization": _SUBSCRIPTION_OAUTH_CREDENTIAL},
|
|
),
|
|
]
|
|
|
|
|
|
def _run_anthropic_hop_with_shared_headers(shared_headers):
|
|
litellm.completion(
|
|
model="anthropic/claude-3-5-sonnet-20240620",
|
|
messages=[{"role": "user", "content": "Say OK"}],
|
|
extra_headers=shared_headers,
|
|
provider_specific_header=_scoped_headers_for_oauth_request(),
|
|
api_key="sk-fake-anthropic-key",
|
|
mock_response="OK",
|
|
)
|
|
|
|
|
|
def test_completion_does_not_mutate_caller_supplied_headers():
|
|
shared_headers = {"x-tenant": "acme"}
|
|
|
|
_run_anthropic_hop_with_shared_headers(shared_headers)
|
|
|
|
assert shared_headers == {"x-tenant": "acme"}
|
|
|
|
|
|
def test_anthropic_oauth_credential_does_not_persist_into_next_provider_hop():
|
|
shared_headers = {"x-tenant": "acme"}
|
|
|
|
_run_anthropic_hop_with_shared_headers(shared_headers)
|
|
|
|
leaked = [name for name, value in shared_headers.items() if value == _SUBSCRIPTION_OAUTH_CREDENTIAL]
|
|
assert leaked == []
|
|
assert "anthropic-version" not in shared_headers
|
|
|
|
|
|
STREAM_COST_MODEL = "gpt-4o"
|
|
STREAMED_USAGE = {"prompt_tokens": 137, "completion_tokens": 42, "total_tokens": 179}
|
|
|
|
|
|
def _text_chunk(content, finish_reason=None, usage=None):
|
|
chunk = {
|
|
"id": "chatcmpl-stream-cost",
|
|
"object": "chat.completion.chunk",
|
|
"created": 1700000000,
|
|
"model": STREAM_COST_MODEL,
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"delta": {"role": "assistant", "content": content},
|
|
"finish_reason": finish_reason,
|
|
}
|
|
],
|
|
}
|
|
if usage is not None:
|
|
chunk["usage"] = usage
|
|
return chunk
|
|
|
|
|
|
def _priced_at(prompt_tokens, completion_tokens):
|
|
prices = litellm.model_cost[STREAM_COST_MODEL]
|
|
return (
|
|
prompt_tokens * prices["input_cost_per_token"]
|
|
+ completion_tokens * prices["output_cost_per_token"]
|
|
)
|
|
|
|
|
|
@pytest.fixture
|
|
def local_cost_map(monkeypatch):
|
|
"""The prices these tests assert are the checked-in ones. Setting the environment
|
|
variable alone does not reload the map, so pin the map itself.
|
|
|
|
``get_model_info`` is lru_cached, so pinning ``model_cost`` is not enough on its
|
|
own: a cached entry warmed against the network-fetched map keeps its old prices
|
|
and ``completion_cost`` bills at those while the assertions read the pinned map.
|
|
Clear on the way in and out so entries never leak across tests in either direction."""
|
|
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
|
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
|
litellm.get_model_info.cache_clear()
|
|
yield
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_a_streamed_response_bills_the_usage_the_provider_reported(local_cost_map):
|
|
rebuilt = litellm.stream_chunk_builder(
|
|
chunks=[
|
|
_text_chunk("Hello"),
|
|
_text_chunk(" there"),
|
|
_text_chunk(None, finish_reason="stop", usage=STREAMED_USAGE),
|
|
],
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
)
|
|
|
|
assert rebuilt.choices[0].message.content == "Hello there"
|
|
assert rebuilt.usage.prompt_tokens == STREAMED_USAGE["prompt_tokens"]
|
|
assert rebuilt.usage.completion_tokens == STREAMED_USAGE["completion_tokens"]
|
|
|
|
cost = litellm.completion_cost(completion_response=rebuilt, model=STREAM_COST_MODEL)
|
|
|
|
assert cost == pytest.approx(_priced_at(137, 42))
|
|
assert cost == pytest.approx(0.0007625)
|
|
|
|
|
|
def test_streaming_and_not_streaming_bill_the_same_usage_the_same(local_cost_map):
|
|
rebuilt = litellm.stream_chunk_builder(
|
|
chunks=[
|
|
_text_chunk("Hello"),
|
|
_text_chunk(" there"),
|
|
_text_chunk(None, finish_reason="stop", usage=STREAMED_USAGE),
|
|
],
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
)
|
|
whole = litellm.ModelResponse(
|
|
id="chatcmpl-stream-cost",
|
|
model=STREAM_COST_MODEL,
|
|
object="chat.completion",
|
|
created=1700000000,
|
|
choices=[
|
|
{
|
|
"index": 0,
|
|
"message": {"role": "assistant", "content": "Hello there"},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
usage=STREAMED_USAGE,
|
|
)
|
|
|
|
assert litellm.completion_cost(
|
|
completion_response=rebuilt, model=STREAM_COST_MODEL
|
|
) == pytest.approx(litellm.completion_cost(completion_response=whole, model=STREAM_COST_MODEL))
|
|
|
|
|
|
def test_a_stream_that_reported_no_usage_is_still_billed(local_cost_map):
|
|
rebuilt = litellm.stream_chunk_builder(
|
|
chunks=[
|
|
_text_chunk("Hello"),
|
|
_text_chunk(" there"),
|
|
_text_chunk(None, finish_reason="stop"),
|
|
],
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
)
|
|
|
|
assert rebuilt.usage.prompt_tokens > 0
|
|
assert rebuilt.usage.completion_tokens > 0
|
|
|
|
cost = litellm.completion_cost(completion_response=rebuilt, model=STREAM_COST_MODEL)
|
|
|
|
assert cost > 0
|
|
assert cost == pytest.approx(
|
|
_priced_at(rebuilt.usage.prompt_tokens, rebuilt.usage.completion_tokens)
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_acompletion_resolves_provider_from_api_base():
|
|
response = await litellm.acompletion(
|
|
model="deepseek-chat",
|
|
api_base="https://api.deepseek.com/v1",
|
|
api_key="fake-key",
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
mock_response="resolved",
|
|
)
|
|
|
|
assert response.choices[0].message.content == "resolved"
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class _RecordedSpeechSuccess:
|
|
call_type: str | None
|
|
spend_metadata: Mapping[str, object]
|
|
response_cost: float | None
|
|
logged_response_cost: float | None
|
|
|
|
|
|
def _record_speech_success(payload: dict[str, object]) -> _RecordedSpeechSuccess:
|
|
call_type: Final = payload.get("call_type")
|
|
response_cost: Final = payload.get("response_cost")
|
|
logging_payload: Final = payload.get("standard_logging_object")
|
|
logged_cost: Final = logging_payload.get("response_cost") if isinstance(logging_payload, dict) else None
|
|
return _RecordedSpeechSuccess(
|
|
call_type=call_type if isinstance(call_type, str) else None,
|
|
spend_metadata=get_litellm_metadata_from_kwargs(payload),
|
|
response_cost=response_cost if isinstance(response_cost, float) else None,
|
|
logged_response_cost=logged_cost if isinstance(logged_cost, float) else None,
|
|
)
|
|
|
|
|
|
class _SuccessEventRecorder(CustomLogger):
|
|
def __init__(self) -> None:
|
|
super().__init__()
|
|
self.events: list[_RecordedSpeechSuccess] = [] # mutable-ok: test recorder of success-callback events
|
|
|
|
async def async_log_success_event(
|
|
self, kwargs: dict[str, object], response_obj: object, start_time: object, end_time: object
|
|
) -> None:
|
|
self.events.append(_record_speech_success(kwargs))
|
|
|
|
|
|
async def _wait_for_success_event(recorder: _SuccessEventRecorder, call_type: str) -> _RecordedSpeechSuccess:
|
|
for _ in range(100):
|
|
if (event := next((e for e in recorder.events if e.call_type == call_type), None)) is not None:
|
|
return event
|
|
await asyncio.sleep(0.05)
|
|
pytest.fail(f"no {call_type} success event; got {[e.call_type for e in recorder.events]}")
|
|
|
|
|
|
def _gemini_tts_generate_content_response() -> dict[str, object]:
|
|
return {
|
|
"candidates": [
|
|
{
|
|
"content": {
|
|
"parts": [
|
|
{
|
|
"inlineData": {
|
|
"mimeType": "audio/L16;codec=pcm;rate=24000",
|
|
"data": base64.b64encode(b"pcm-audio-bytes").decode(),
|
|
}
|
|
}
|
|
],
|
|
"role": "model",
|
|
},
|
|
"finishReason": "STOP",
|
|
"index": 0,
|
|
}
|
|
],
|
|
"usageMetadata": {
|
|
"promptTokenCount": 5,
|
|
"candidatesTokenCount": 60,
|
|
"totalTokenCount": 65,
|
|
"promptTokensDetails": [{"modality": "TEXT", "tokenCount": 5}],
|
|
"candidatesTokensDetails": [{"modality": "AUDIO", "tokenCount": 60}],
|
|
},
|
|
"modelVersion": "gemini-2.5-flash-preview-tts",
|
|
}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_aspeech_gemini_bridge_keeps_proxy_metadata_for_spend_tracking(
|
|
respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
|
monkeypatch.delenv("GEMINI_API_KEY", raising=False)
|
|
monkeypatch.delenv("GOOGLE_API_KEY", raising=False)
|
|
recorder: Final = _SuccessEventRecorder()
|
|
monkeypatch.setattr(litellm, "callbacks", [recorder])
|
|
mock_route: Final = respx_mock.post(
|
|
url__regex=r"https://generativelanguage\.googleapis\.com/v1beta/models/gemini-2\.5-flash-preview-tts:generateContent.*"
|
|
).mock(return_value=httpx.Response(200, json=_gemini_tts_generate_content_response()))
|
|
|
|
await litellm.aspeech(
|
|
model="gemini/gemini-2.5-flash-preview-tts",
|
|
input="spend tracking check",
|
|
voice="Kore",
|
|
api_key="fake-gemini-key",
|
|
metadata={"user_api_key": "hashed-virtual-key", "user_api_key_user_id": "user-1"},
|
|
)
|
|
|
|
assert mock_route.called
|
|
assert mock_route.calls.last.request.headers["x-goog-api-key"] == "fake-gemini-key"
|
|
speech_event: Final = await _wait_for_success_event(recorder, call_type="aspeech")
|
|
assert speech_event.spend_metadata["user_api_key"] == "hashed-virtual-key"
|
|
assert speech_event.spend_metadata["user_api_key_user_id"] == "user-1"
|
|
expected_prompt_cost, expected_completion_cost = litellm.cost_per_token(
|
|
model="gemini/gemini-2.5-flash-preview-tts",
|
|
usage_object=Usage(prompt_tokens=5, completion_tokens=60, total_tokens=65),
|
|
)
|
|
expected_cost: Final = expected_prompt_cost + expected_completion_cost
|
|
assert expected_cost > 0
|
|
assert speech_event.response_cost == pytest.approx(expected_cost)
|
|
assert speech_event.logged_response_cost == pytest.approx(expected_cost)
|