mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
The rebuild's tool-call selection and its text-only fast path only looked at choice 0 of each chunk, so a chunk that packs several choices (Gemini with candidateCount above 1) lost a tool call carried by a later candidate, and a chunk whose later choice had no tool calls at all made the rebuild raise. Both now consider every choice in the chunk.
3859 lines
143 KiB
Python
3859 lines
143 KiB
Python
import asyncio
|
|
import base64
|
|
from datetime import datetime
|
|
import contextlib
|
|
import copy
|
|
import json
|
|
import os
|
|
from collections.abc import Mapping
|
|
from dataclasses import dataclass
|
|
from typing import Final
|
|
|
|
import httpx
|
|
import pytest
|
|
import respx
|
|
from fastapi.testclient import TestClient
|
|
|
|
|
|
import urllib.parse
|
|
from importlib import import_module
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
import litellm
|
|
from litellm import main as litellm_main
|
|
from litellm.integrations.custom_logger import CustomLogger
|
|
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs
|
|
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLogging
|
|
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices, Usage
|
|
|
|
|
|
async def _async_fake_bedrock_image_details(image_url):
|
|
return "ZmFrZS1pbWFnZQ==", "image/png"
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def clear_client_cache():
|
|
"""
|
|
Clear the HTTP client cache before each test to ensure mocks are used.
|
|
This prevents cached real clients from being reused across tests.
|
|
"""
|
|
cache = getattr(litellm, "in_memory_llm_clients_cache", None)
|
|
if cache is not None:
|
|
cache.flush_cache()
|
|
yield
|
|
if cache is not None:
|
|
cache.flush_cache()
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def add_api_keys_to_env(monkeypatch):
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-api03-1234567890")
|
|
monkeypatch.setenv("OPENAI_API_KEY", "sk-openai-api03-1234567890")
|
|
monkeypatch.setenv("AWS_ACCESS_KEY_ID", "my-fake-aws-access-key-id")
|
|
monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "my-fake-aws-secret-access-key")
|
|
monkeypatch.setenv("AWS_REGION", "us-east-1")
|
|
# Keep these transformation tests on the simple access-key path. A leaked
|
|
# session token or role/web-identity env var pushes Bedrock auth down a
|
|
# different branch and fails before the mocked HTTP client is exercised.
|
|
monkeypatch.delenv("AWS_SESSION_TOKEN", raising=False)
|
|
monkeypatch.delenv("AWS_ROLE_ARN", raising=False)
|
|
monkeypatch.delenv("AWS_WEB_IDENTITY_TOKEN_FILE", raising=False)
|
|
|
|
|
|
@pytest.fixture
|
|
def openai_api_response():
|
|
mock_response_data = {
|
|
"id": "chatcmpl-B0W3vmiM78Xkgx7kI7dr7PC949DMS",
|
|
"choices": [
|
|
{
|
|
"finish_reason": "stop",
|
|
"index": 0,
|
|
"logprobs": None,
|
|
"message": {
|
|
"content": "",
|
|
"refusal": None,
|
|
"role": "assistant",
|
|
"audio": None,
|
|
"function_call": None,
|
|
"tool_calls": None,
|
|
},
|
|
}
|
|
],
|
|
"created": 1739462947,
|
|
"model": "gpt-4o-mini-2024-07-18",
|
|
"object": "chat.completion",
|
|
"service_tier": "default",
|
|
"system_fingerprint": "fp_bd83329f63",
|
|
"usage": {
|
|
"completion_tokens": 1,
|
|
"prompt_tokens": 121,
|
|
"total_tokens": 122,
|
|
"completion_tokens_details": {
|
|
"accepted_prediction_tokens": 0,
|
|
"audio_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"rejected_prediction_tokens": 0,
|
|
},
|
|
"prompt_tokens_details": {"audio_tokens": 0, "cached_tokens": 0},
|
|
},
|
|
}
|
|
|
|
return mock_response_data
|
|
|
|
|
|
def test_completion_missing_role(openai_api_response):
|
|
from openai import OpenAI
|
|
|
|
from litellm.types.utils import ModelResponse
|
|
|
|
client = OpenAI(api_key="test_api_key")
|
|
|
|
mock_raw_response = MagicMock()
|
|
mock_raw_response.headers = {
|
|
"x-request-id": "123",
|
|
"openai-organization": "org-123",
|
|
"x-ratelimit-limit-requests": "100",
|
|
"x-ratelimit-remaining-requests": "99",
|
|
}
|
|
mock_raw_response.parse.return_value = ModelResponse(**openai_api_response)
|
|
|
|
print(f"openai_api_response: {openai_api_response}")
|
|
|
|
with patch.object(
|
|
client.chat.completions.with_raw_response, "create", MagicMock(return_value=mock_raw_response)
|
|
) as mock_create:
|
|
litellm.completion(
|
|
model="gpt-4o-mini",
|
|
messages=[
|
|
{"role": "user", "content": "Hey"},
|
|
{
|
|
"content": "",
|
|
"tool_calls": [
|
|
{
|
|
"id": "call_m0vFJjQmTH1McvaHBPR2YFwY",
|
|
"function": {
|
|
"arguments": '{"input": "dksjsdkjdhskdjshdskhjkhlk"}',
|
|
"name": "tool_name",
|
|
},
|
|
"type": "function",
|
|
"index": 0,
|
|
},
|
|
{
|
|
"id": "call_Vw6RaqV2n5aaANXEdp5pYxo2",
|
|
"function": {
|
|
"arguments": '{"input": "jkljlkjlkjlkjlk"}',
|
|
"name": "tool_name",
|
|
},
|
|
"type": "function",
|
|
"index": 1,
|
|
},
|
|
{
|
|
"id": "call_hBIKwldUEGlNh6NlSXil62K4",
|
|
"function": {
|
|
"arguments": '{"input": "jkjlkjlkjlkj;lj"}',
|
|
"name": "tool_name",
|
|
},
|
|
"type": "function",
|
|
"index": 2,
|
|
},
|
|
],
|
|
},
|
|
],
|
|
client=client,
|
|
)
|
|
|
|
mock_create.assert_called_once()
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"model",
|
|
[
|
|
"gemini/gemini-1.5-flash",
|
|
"bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
"bedrock/invoke/anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
"anthropic/claude-3-5-sonnet",
|
|
],
|
|
)
|
|
@pytest.mark.parametrize("sync_mode", [True, False])
|
|
@pytest.mark.asyncio
|
|
async def test_url_with_format_param(model, sync_mode, monkeypatch):
|
|
from litellm import acompletion, completion
|
|
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
|
from litellm.litellm_core_utils.prompt_templates import factory as prompt_factory
|
|
|
|
if sync_mode:
|
|
client = HTTPHandler()
|
|
else:
|
|
client = AsyncHTTPHandler()
|
|
|
|
# This test is about request shaping, not live image downloads. Stub the
|
|
# URL->image conversion helpers so suite-level network/client state from
|
|
# earlier tests cannot prevent the mocked provider client from being hit.
|
|
fake_base64_image = "data:image/png;base64,ZmFrZS1pbWFnZQ=="
|
|
monkeypatch.setattr(
|
|
prompt_factory, "convert_url_to_base64", lambda url: fake_base64_image
|
|
)
|
|
monkeypatch.setattr(
|
|
prompt_factory.BedrockImageProcessor,
|
|
"get_image_details",
|
|
staticmethod(lambda image_url: ("ZmFrZS1pbWFnZQ==", "image/png")),
|
|
)
|
|
monkeypatch.setattr(
|
|
prompt_factory.BedrockImageProcessor,
|
|
"get_image_details_async",
|
|
staticmethod(_async_fake_bedrock_image_details),
|
|
)
|
|
|
|
args = {
|
|
"model": model,
|
|
"messages": [
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{
|
|
"type": "image_url",
|
|
"image_url": {
|
|
"url": "https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/c233c9ade2ccb5491072ae232c814942.png",
|
|
"format": "image/png",
|
|
},
|
|
},
|
|
{"type": "text", "text": "Describe this image"},
|
|
],
|
|
}
|
|
],
|
|
}
|
|
if model.startswith("gemini/"):
|
|
args["api_key"] = "test-api-key"
|
|
with patch.object(client, "post", new=MagicMock()) as mock_client:
|
|
try:
|
|
if sync_mode:
|
|
response = completion(**args, client=client)
|
|
else:
|
|
response = await acompletion(**args, client=client)
|
|
print(response)
|
|
except Exception as e:
|
|
pass
|
|
|
|
mock_client.assert_called()
|
|
|
|
print(mock_client.call_args.kwargs)
|
|
|
|
if "data" in mock_client.call_args.kwargs:
|
|
json_str = mock_client.call_args.kwargs["data"]
|
|
else:
|
|
json_str = json.dumps(mock_client.call_args.kwargs["json"])
|
|
|
|
if isinstance(json_str, bytes):
|
|
json_str = json_str.decode("utf-8")
|
|
|
|
print(f"type of json_str: {type(json_str)}")
|
|
|
|
# Bedrock models convert URLs to base64, while direct Anthropic models support URLs
|
|
# bedrock/invoke models use Anthropic messages API which supports URLs
|
|
if model.startswith("bedrock/invoke/"):
|
|
# bedrock/invoke should convert URLs to base64 (doesn't support URL references)
|
|
# URL should NOT be in the JSON (it should be converted to base64)
|
|
assert "https://awsmp-logos.s3.amazonaws.com" not in json_str
|
|
# Should have base64 data in the source (type="base64", not type="url")
|
|
assert '"type":"base64"' in json_str or '"type": "base64"' in json_str
|
|
# Should have "data" field containing base64 content
|
|
assert '"data"' in json_str
|
|
elif model.startswith("bedrock/"):
|
|
# Regular Bedrock models should convert URLs to base64 (uses "bytes" field)
|
|
# URL should NOT be in the JSON (it should be converted to base64)
|
|
assert "https://awsmp-logos.s3.amazonaws.com" not in json_str
|
|
# Should have "bytes" field (Bedrock uses "bytes" not "base64" in the field name)
|
|
assert '"bytes"' in json_str or '"bytes":' in json_str
|
|
elif model.startswith("anthropic/"):
|
|
# Direct Anthropic models should pass HTTPS URLs directly (HTTP URLs are converted to base64)
|
|
# Since we're using HTTPS URL, it should be passed as-is
|
|
assert "https://awsmp-logos.s3.amazonaws.com" in json_str
|
|
# For Anthropic, URL references use "url" type, not base64
|
|
assert '"type":"url"' in json_str or '"type": "url"' in json_str
|
|
else:
|
|
# For other models, check format parameter is respected
|
|
assert "png" in json_str
|
|
assert "jpeg" not in json_str
|
|
|
|
|
|
@pytest.mark.parametrize("model", ["gpt-4o-mini"])
|
|
@pytest.mark.parametrize("sync_mode", [True, False])
|
|
@pytest.mark.asyncio
|
|
async def test_url_with_format_param_openai(model, sync_mode):
|
|
from openai import AsyncOpenAI, OpenAI
|
|
|
|
from litellm import acompletion, completion
|
|
|
|
if sync_mode:
|
|
client = OpenAI()
|
|
else:
|
|
client = AsyncOpenAI()
|
|
|
|
args = {
|
|
"model": model,
|
|
"messages": [
|
|
{
|
|
"role": "user",
|
|
"content": [
|
|
{
|
|
"type": "image_url",
|
|
"image_url": {
|
|
"url": "https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/c233c9ade2ccb5491072ae232c814942.png",
|
|
"format": "image/png",
|
|
},
|
|
},
|
|
{"type": "text", "text": "Describe this image"},
|
|
],
|
|
}
|
|
],
|
|
}
|
|
with patch.object(
|
|
client.chat.completions.with_raw_response, "create"
|
|
) as mock_client:
|
|
try:
|
|
if sync_mode:
|
|
response = completion(**args, client=client)
|
|
else:
|
|
response = await acompletion(**args, client=client)
|
|
print(response)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_client.assert_called()
|
|
|
|
print(mock_client.call_args.kwargs)
|
|
|
|
json_str = json.dumps(mock_client.call_args.kwargs)
|
|
|
|
assert "format" not in json_str
|
|
|
|
|
|
def test_bedrock_latency_optimized_inference():
|
|
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
|
|
|
client = HTTPHandler()
|
|
with patch.object(client, "post") as mock_post:
|
|
try:
|
|
response = litellm.completion(
|
|
model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
|
performanceConfig={"latency": "optimized"},
|
|
client=client,
|
|
)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_post.assert_called_once()
|
|
json_data = json.loads(mock_post.call_args.kwargs["data"])
|
|
assert json_data["performanceConfig"]["latency"] == "optimized"
|
|
|
|
|
|
def test_strip_input_examples_for_non_anthropic_providers():
|
|
tools = [
|
|
{
|
|
"type": "function",
|
|
"name": "example_tool",
|
|
"input_examples": [{"foo": "bar"}],
|
|
"function": {
|
|
"name": "example_tool",
|
|
"input_examples": [{"foo": "bar"}],
|
|
},
|
|
}
|
|
]
|
|
|
|
assert not litellm_main._should_allow_input_examples(
|
|
custom_llm_provider="openai", model="gpt-4o-mini"
|
|
)
|
|
|
|
cleaned = litellm_main._drop_input_examples_from_tools(tools=tools)
|
|
|
|
assert isinstance(cleaned, list)
|
|
assert "input_examples" not in cleaned[0]
|
|
assert "input_examples" not in cleaned[0]["function"]
|
|
|
|
|
|
def test_custom_provider_with_extra_headers():
|
|
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
|
|
|
with patch.object(
|
|
litellm.llms.custom_httpx.http_handler.HTTPHandler, "post"
|
|
) as mock_post:
|
|
response = litellm.completion(
|
|
model="custom/custom",
|
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
|
headers={"X-Custom-Header": "custom-value"},
|
|
api_base="https://example.com/api/v1",
|
|
)
|
|
|
|
mock_post.assert_called_once()
|
|
assert mock_post.call_args[1]["headers"]["X-Custom-Header"] == "custom-value"
|
|
|
|
|
|
def test_custom_provider_with_extra_body():
|
|
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
|
|
|
with patch.object(
|
|
litellm.llms.custom_httpx.http_handler.HTTPHandler, "post"
|
|
) as mock_post:
|
|
response = litellm.completion(
|
|
model="custom/custom",
|
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
|
extra_body={
|
|
"X-Custom-BodyValue": "custom-value",
|
|
"X-Custom-BodyValue2": "custom-value2",
|
|
},
|
|
api_base="https://example.com/api/v1",
|
|
)
|
|
mock_post.assert_called_once()
|
|
|
|
assert mock_post.call_args[1]["json"]["X-Custom-BodyValue"] == "custom-value"
|
|
assert mock_post.call_args[1]["json"] == {
|
|
"model": "custom",
|
|
"params": {
|
|
"prompt": ["Hello, how are you?"],
|
|
"max_tokens": None,
|
|
"temperature": None,
|
|
"top_p": None,
|
|
"top_k": None,
|
|
},
|
|
"X-Custom-BodyValue": "custom-value",
|
|
"X-Custom-BodyValue2": "custom-value2",
|
|
}
|
|
|
|
# test that extra_body is not passed if not provided
|
|
with patch.object(
|
|
litellm.llms.custom_httpx.http_handler.HTTPHandler, "post"
|
|
) as mock_post:
|
|
response = litellm.completion(
|
|
model="custom/custom",
|
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
|
api_base="https://example.com/api/v1",
|
|
)
|
|
mock_post.assert_called_once()
|
|
assert mock_post.call_args[1]["json"] == {
|
|
"model": "custom",
|
|
"params": {
|
|
"prompt": ["Hello, how are you?"],
|
|
"max_tokens": None,
|
|
"temperature": None,
|
|
"top_p": None,
|
|
"top_k": None,
|
|
},
|
|
}
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def set_openrouter_api_key():
|
|
original_api_key = os.environ.get("OPENROUTER_API_KEY")
|
|
os.environ["OPENROUTER_API_KEY"] = "fake-key-for-testing"
|
|
yield
|
|
if original_api_key is not None:
|
|
os.environ["OPENROUTER_API_KEY"] = original_api_key
|
|
else:
|
|
del os.environ["OPENROUTER_API_KEY"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_extra_body_with_fallback(
|
|
respx_mock: respx.MockRouter, set_openrouter_api_key, monkeypatch
|
|
):
|
|
"""
|
|
test regression for https://github.com/BerriAI/litellm/issues/8425.
|
|
|
|
This was perhaps a wider issue with the acompletion function not passing kwargs such as extra_body correctly when fallbacks are specified.
|
|
"""
|
|
|
|
# Save original state to restore after test
|
|
original_disable_aiohttp = litellm.disable_aiohttp_transport
|
|
|
|
try:
|
|
# since this uses respx, we need to set use_aiohttp_transport to False
|
|
# Set both the global variable and environment variable to ensure it takes effect
|
|
litellm.disable_aiohttp_transport = True
|
|
monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True")
|
|
# Flush cache to ensure no stale aiohttp clients are used
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
# Set up test parameters
|
|
model = "openrouter/deepseek/deepseek-chat"
|
|
messages = [{"role": "user", "content": "Hello, world!"}]
|
|
extra_body = {
|
|
"provider": {
|
|
"order": ["DeepSeek"],
|
|
"allow_fallbacks": False,
|
|
"require_parameters": True,
|
|
}
|
|
}
|
|
fallbacks = [{"model": "openrouter/google/gemini-flash-1.5-8b"}]
|
|
|
|
# Set up mock to respond to any POST request to the OpenRouter endpoint
|
|
# This ensures it works for both primary and fallback models
|
|
mock_route = respx_mock.post("https://openrouter.ai/api/v1/chat/completions")
|
|
mock_route.return_value = httpx.Response(
|
|
200,
|
|
json={
|
|
"id": "chatcmpl-123",
|
|
"object": "chat.completion",
|
|
"created": 1677652288,
|
|
"model": model,
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "Hello from mocked response!",
|
|
},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
"usage": {
|
|
"prompt_tokens": 9,
|
|
"completion_tokens": 12,
|
|
"total_tokens": 21,
|
|
},
|
|
},
|
|
)
|
|
|
|
response = await litellm.acompletion(
|
|
model=model,
|
|
messages=messages,
|
|
extra_body=extra_body,
|
|
fallbacks=fallbacks,
|
|
api_key="fake-openrouter-api-key",
|
|
)
|
|
|
|
# Verify the response
|
|
assert response is not None
|
|
assert (
|
|
len(respx_mock.calls) > 0
|
|
), "Mock was not called - check if aiohttp transport is properly disabled"
|
|
|
|
# Get the request from the mock
|
|
request: httpx.Request = respx_mock.calls[0].request
|
|
request_body = request.read()
|
|
request_body = json.loads(request_body)
|
|
|
|
# Verify basic parameters
|
|
assert request_body["model"] == "deepseek/deepseek-chat"
|
|
assert request_body["messages"] == messages
|
|
|
|
# Verify the extra_body parameters remain under the provider key
|
|
assert request_body["provider"]["order"] == ["DeepSeek"]
|
|
assert request_body["provider"]["allow_fallbacks"] is False
|
|
assert request_body["provider"]["require_parameters"] is True
|
|
finally:
|
|
# Restore original state to prevent test pollution
|
|
litellm.disable_aiohttp_transport = original_disable_aiohttp
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
|
|
@pytest.mark.parametrize("env_base", ["OPENAI_BASE_URL", "OPENAI_API_BASE"])
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.flaky(retries=3, delay=1)
|
|
async def test_openai_env_base(
|
|
respx_mock: respx.MockRouter, env_base, openai_api_response, monkeypatch
|
|
):
|
|
"This tests OpenAI env variables are honored, including legacy OPENAI_API_BASE"
|
|
# Ensure aiohttp transport is disabled to use httpx which respx can mock
|
|
litellm.disable_aiohttp_transport = True
|
|
|
|
expected_base_url = "http://localhost:12345/v1"
|
|
|
|
# Assign the environment variable based on env_base, and use a fake API key.
|
|
monkeypatch.setenv(env_base, expected_base_url)
|
|
monkeypatch.setenv("OPENAI_API_KEY", "fake_openai_api_key")
|
|
|
|
model = "gpt-4o"
|
|
messages = [{"role": "user", "content": "Hello, how are you?"}]
|
|
|
|
# Configure respx mock to intercept the request
|
|
mock_route = respx_mock.post(
|
|
url__regex=r"http://localhost:12345/v1/chat/completions.*"
|
|
).mock(
|
|
return_value=httpx.Response(
|
|
status_code=200,
|
|
json={
|
|
"id": "chatcmpl-123",
|
|
"object": "chat.completion",
|
|
"created": 1677652288,
|
|
"model": model,
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "Hello from mocked response!",
|
|
},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
"usage": {
|
|
"prompt_tokens": 9,
|
|
"completion_tokens": 12,
|
|
"total_tokens": 21,
|
|
},
|
|
},
|
|
)
|
|
)
|
|
|
|
try:
|
|
response = await litellm.acompletion(model=model, messages=messages)
|
|
|
|
# verify we had a response
|
|
assert response.choices[0].message.content == "Hello from mocked response!"
|
|
|
|
# Verify the mock was called
|
|
assert (
|
|
mock_route.called
|
|
), "Mock route was not called - request may have bypassed respx"
|
|
finally:
|
|
# Clean up to avoid affecting other tests
|
|
litellm.disable_aiohttp_transport = False
|
|
|
|
|
|
def build_database_url(username, password, host, dbname):
|
|
username_enc = urllib.parse.quote_plus(username)
|
|
password_enc = urllib.parse.quote_plus(password)
|
|
dbname_enc = urllib.parse.quote_plus(dbname)
|
|
return f"postgresql://{username_enc}:{password_enc}@{host}/{dbname_enc}"
|
|
|
|
|
|
def test_build_database_url():
|
|
url = build_database_url("user@name", "p@ss:word", "localhost", "db/name")
|
|
assert url == "postgresql://user%40name:p%40ss%3Aword@localhost/db%2Fname"
|
|
|
|
|
|
def test_bedrock_llama():
|
|
litellm._turn_on_debug()
|
|
from litellm.types.utils import CallTypes
|
|
from litellm.utils import return_raw_request
|
|
|
|
model = "bedrock/invoke/us.meta.llama4-scout-17b-instruct-v1:0"
|
|
|
|
request = return_raw_request(
|
|
endpoint=CallTypes.completion,
|
|
kwargs={
|
|
"model": model,
|
|
"messages": [
|
|
{"role": "user", "content": "hi"},
|
|
],
|
|
},
|
|
)
|
|
print(request)
|
|
|
|
assert (
|
|
request["raw_request_body"]["prompt"]
|
|
== "<|begin_of_text|><|start_header_id|>user<|end_header_id|>\n\nhi<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n"
|
|
)
|
|
|
|
|
|
def _mocked_openai_chat_response(model: str) -> httpx.Response:
|
|
return httpx.Response(
|
|
status_code=200,
|
|
json={
|
|
"id": "chatcmpl-123",
|
|
"object": "chat.completion",
|
|
"created": 1677652288,
|
|
"model": model,
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"message": {
|
|
"role": "assistant",
|
|
"content": "Hello from mocked response!",
|
|
},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
"usage": {
|
|
"prompt_tokens": 9,
|
|
"completion_tokens": 12,
|
|
"total_tokens": 21,
|
|
},
|
|
},
|
|
)
|
|
|
|
|
|
def test_completion_forwards_verbosity_in_raw_request(respx_mock: respx.MockRouter):
|
|
"""Regression test: completion() must forward the verbosity param to the provider request body."""
|
|
from litellm.types.utils import CallTypes
|
|
from litellm.utils import return_raw_request
|
|
|
|
model = "gpt-5.2"
|
|
messages = [{"role": "user", "content": "hi"}]
|
|
respx_mock.post("https://api.openai.com/v1/chat/completions").mock(
|
|
return_value=_mocked_openai_chat_response(model)
|
|
)
|
|
|
|
request = return_raw_request(
|
|
endpoint=CallTypes.completion,
|
|
kwargs={
|
|
"model": model,
|
|
"messages": messages,
|
|
"verbosity": "high",
|
|
},
|
|
)
|
|
|
|
assert request["raw_request_body"]["verbosity"] == "high"
|
|
assert request["raw_request_body"]["model"] == model
|
|
assert request["raw_request_body"]["messages"] == messages
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_acompletion_forwards_verbosity_to_provider_request(
|
|
respx_mock: respx.MockRouter, monkeypatch
|
|
):
|
|
"""Regression test: acompletion() must forward the verbosity param to the provider request body."""
|
|
original_disable_aiohttp = litellm.disable_aiohttp_transport
|
|
try:
|
|
litellm.disable_aiohttp_transport = True
|
|
monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True")
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
model = "gpt-5.2"
|
|
messages = [{"role": "user", "content": "hi"}]
|
|
mock_route = respx_mock.post("https://api.openai.com/v1/chat/completions").mock(
|
|
return_value=_mocked_openai_chat_response(model)
|
|
)
|
|
|
|
response = await litellm.acompletion(
|
|
model=model,
|
|
messages=messages,
|
|
verbosity="low",
|
|
api_key="fake-openai-api-key",
|
|
)
|
|
|
|
assert response.choices[0].message.content == "Hello from mocked response!"
|
|
assert mock_route.called
|
|
request_body = json.loads(respx_mock.calls[0].request.read())
|
|
assert request_body["verbosity"] == "low"
|
|
assert request_body["model"] == model
|
|
assert request_body["messages"] == messages
|
|
finally:
|
|
litellm.disable_aiohttp_transport = original_disable_aiohttp
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
|
|
def test_responses_api_bridge_check_strips_responses_prefix():
|
|
"""Test that responses_api_bridge_check strips 'responses/' prefix and sets mode."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 4096}
|
|
|
|
model_info, model = responses_api_bridge_check(
|
|
model="responses/gpt-4-responses",
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model == "gpt-4-responses"
|
|
assert model_info["mode"] == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_pro():
|
|
"""Test that gpt-5.4-pro routes through responses API bridge, not chat completions.
|
|
|
|
Regression test for https://github.com/BerriAI/litellm/issues/23014
|
|
gpt-5.4-pro is a responses-only model and must not be sent to /v1/chat/completions.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
for model_name in ["gpt-5.4-pro", "gpt-5.4-pro-2026-03-05"]:
|
|
model_info, model = responses_api_bridge_check(
|
|
model=model_name,
|
|
custom_llm_provider="openai",
|
|
)
|
|
assert (
|
|
model_info.get("mode") == "responses"
|
|
), f"{model_name} should have mode='responses', got '{model_info.get('mode')}'"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_tools_plus_reasoning_routes_to_responses():
|
|
"""gpt-5.4 with both tools and reasoning_effort should route to Responses API."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="xhigh",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_6_astra_tools_with_default_reasoning_routes_to_responses():
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-6-astra",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
)
|
|
|
|
assert model == "gpt-6-astra"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_5_tools_plus_reasoning_routes_to_responses():
|
|
"""gpt-5.5+ with both tools and reasoning_effort should route to Responses API."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.5-pro",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="xhigh",
|
|
)
|
|
|
|
assert model == "gpt-5.5-pro"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_azure_gpt_5_4_tools_plus_reasoning_routes_to_responses():
|
|
"""Azure gpt-5.4 with both tools and reasoning_effort should route to Responses API."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="azure",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="high",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_azure_gpt_5_4_tools_with_default_reasoning_routes_to_responses():
|
|
"""
|
|
Azure gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables
|
|
reasoning by default for gpt-5.4+, and Chat Completions rejects function tools
|
|
whenever reasoning is on.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="azure",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_tools_with_default_reasoning_routes_to_responses():
|
|
"""
|
|
gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables reasoning
|
|
by default for gpt-5.4+, and Chat Completions rejects function tools whenever
|
|
reasoning is on ("use /v1/responses or set reasoning_effort to 'none'").
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"model_name, expected_mode",
|
|
[
|
|
pytest.param("gpt-5.6-sol", "responses", id="above-boundary-bridges"),
|
|
pytest.param("gpt-5.1", None, id="below-boundary-stays-chat"),
|
|
],
|
|
)
|
|
def test_responses_api_bridge_check_gpt_5_6_tools_with_default_reasoning_routes_to_responses(
|
|
monkeypatch, model_name, expected_mode
|
|
):
|
|
"""
|
|
gpt-5.6 must bridge on function tools alone. The bridge used to require an explicit
|
|
reasoning_effort, so a gpt-5.6 call carrying tools and no effort was rejected with
|
|
"Function tools with reasoning_effort are not supported for gpt-5.6-sol in
|
|
/v1/chat/completions".
|
|
|
|
Paired with a model below the gpt-5.4 boundary, which must still stay on chat. The
|
|
gate parses the version and drops any suffix, so the family members bridge
|
|
identically and only the boundary distinguishes behaviour.
|
|
"""
|
|
import litellm
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
|
monkeypatch.setattr(litellm, "api_base", None)
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model=model_name,
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == model_name
|
|
assert model_info.get("mode") == expected_mode
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_tools_with_reasoning_none_stays_chat():
|
|
"""
|
|
Explicit reasoning_effort "none" is OpenAI's documented escape hatch that keeps
|
|
function tools servable on Chat Completions; the bridge must not fire.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="none",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_reasoning_none_with_summary_still_routes_to_responses():
|
|
"""A reasoning summary is Responses-only regardless of effort value."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
reasoning_effort="none",
|
|
reasoning_summary="detailed",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_custom_tools_only_stays_chat():
|
|
"""
|
|
Chat Completions serves custom (grammar) tools natively with reasoning on; only
|
|
FUNCTION tools trigger the OpenAI rejection. Custom-only requests must stay on chat
|
|
so responses keep the native custom tool_call shape instead of the bridge's
|
|
function-shaped mapping.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "custom", "custom": {"name": "ApplyPatch", "description": "V4A patch"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_mixed_function_and_custom_tools_routes_to_responses():
|
|
"""One function tool in the mix is enough to make chat unservable with reasoning on."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[
|
|
{"type": "custom", "custom": {"name": "ApplyPatch"}},
|
|
{"type": "function", "function": {"name": "shell"}},
|
|
],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_flat_function_tool_routes_to_responses():
|
|
"""Responses-style flat function tool defs still count as function tools."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "name": "shell", "parameters": {"type": "object"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_dict_effort_none_stays_chat():
|
|
"""The escape hatch must honor litellm's dict form: {"effort": "none"} means reasoning off."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort={"effort": "none"},
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_dict_effort_active_routes_to_responses():
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort={"effort": "low"},
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_dict_effort_none_with_summary_routes_to_responses():
|
|
"""A summary inside the dict form is Responses-only even when effort is none."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort={"effort": "none", "summary": "concise"},
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
@pytest.mark.parametrize("blank_api_base", [None, "", " ", "\t"])
|
|
def test_responses_api_bridge_check_blank_api_base_is_default_openai(blank_api_base):
|
|
"""
|
|
A blank api_base (None, empty, or whitespace) resolves to the default OpenAI
|
|
endpoint downstream, which enforces the reasoning+tools constraint, so gpt-5.4+
|
|
function-tool requests with unset reasoning_effort must still auto-bridge.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=blank_api_base,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_custom_api_base_with_unset_effort_stays_chat():
|
|
"""
|
|
Chat-only OpenAI-compatible backends registered under the openai provider with a
|
|
custom api_base and gpt-5.4+ model names serve tools-without-reasoning fine and
|
|
have no /responses route; the unset-effort arm must not reroute them.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base="http://vllm.internal:8000/v1",
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_custom_api_base_via_global_with_unset_effort_stays_chat(monkeypatch):
|
|
"""
|
|
A custom base set through the litellm.api_base global (not the call arg) is resolved the
|
|
same way the chat handler resolves it, so the unset-effort arm must not reroute a chat-only
|
|
backend to a /responses route it lacks. Regression guard: the gate previously inspected only
|
|
the call-level api_base and bridged these requests.
|
|
"""
|
|
import litellm
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
monkeypatch.setattr(litellm, "api_base", "http://vllm.internal:8000/v1")
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
@pytest.mark.parametrize("env_var", ["OPENAI_BASE_URL", "OPENAI_API_BASE"])
|
|
def test_responses_api_bridge_check_custom_api_base_via_env_with_unset_effort_stays_chat(monkeypatch, env_var):
|
|
"""
|
|
A custom base set via OPENAI_BASE_URL/OPENAI_API_BASE env is resolved identically to the chat
|
|
handler, so the unset-effort arm leaves the request on chat instead of bridging it.
|
|
"""
|
|
import litellm
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
monkeypatch.setattr(litellm, "api_base", None)
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
|
monkeypatch.setenv(env_var, "http://vllm.internal:8000/v1")
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"api_base",
|
|
[
|
|
"https://southcentralus.privatelink.api.openai.com/v1",
|
|
"https://privatelink.corp.api.openai.com/v1",
|
|
"https://api.openai.com:443/v1",
|
|
"https://api.openai.com/v1/",
|
|
"HTTPS://API.OPENAI.COM/v1",
|
|
],
|
|
)
|
|
def test_responses_api_bridge_check_openai_backed_custom_api_base_with_unset_effort_routes_to_responses(api_base):
|
|
"""
|
|
A custom api_base whose host is api.openai.com or a subdomain of it (a PrivateLink hostname, a
|
|
port-qualified or trailing-slash default) still reaches the real OpenAI backend, which rejects
|
|
function tools with reasoning on Chat Completions, so the unset-effort arm must bridge exactly as
|
|
it does for the literal default URL. Regression guard for GH #39353.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=api_base,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"api_base",
|
|
[
|
|
"https://api.openai.com.evil.example/v1",
|
|
"https://notapi.openai.com/v1",
|
|
"https://gateway.example/v1?upstream=api.openai.com",
|
|
"https://openai.internal.example/api.openai.com/v1",
|
|
],
|
|
)
|
|
def test_responses_api_bridge_check_lookalike_custom_api_base_with_unset_effort_stays_chat(api_base):
|
|
"""Only the host decides: api.openai.com appearing elsewhere in the URL is still a foreign backend."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=api_base,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_privatelink_api_base_via_env_with_unset_effort_routes_to_responses(monkeypatch):
|
|
"""A PrivateLink base set through OPENAI_BASE_URL resolves the way the chat handler's does and still bridges."""
|
|
import litellm
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
monkeypatch.setattr(litellm, "api_base", None)
|
|
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
|
monkeypatch.setenv("OPENAI_BASE_URL", "https://southcentralus.privatelink.api.openai.com/v1")
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base=None,
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_custom_api_base_with_explicit_effort_still_routes():
|
|
"""Explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.6",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="high",
|
|
api_base="http://vllm.internal:8000/v1",
|
|
)
|
|
|
|
assert model == "gpt-5.6"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_azure_with_api_base_and_unset_effort_routes():
|
|
"""Azure OpenAI always sets api_base and does enforce the constraint; keep bridging."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="azure",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
api_base="https://myresource.openai.azure.com",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat():
|
|
"""Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.1",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort=None,
|
|
)
|
|
|
|
assert model == "gpt-5.1"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_4_reasoning_summary_without_tools_routes_to_responses():
|
|
"""gpt-5.4+ with reasoning_effort + reasoningSummary but no tools should bridge (AI SDK)."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="openai",
|
|
tools=None,
|
|
reasoning_effort="medium",
|
|
reasoning_summary="auto",
|
|
)
|
|
|
|
assert model == "gpt-5.4"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_reasoning_summary_routes_to_responses():
|
|
"""Bare ``gpt-5`` with reasoning_effort + reasoningSummary should bridge (not 5.4+)."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5",
|
|
custom_llm_provider="openai",
|
|
tools=None,
|
|
reasoning_effort="medium",
|
|
reasoning_summary="auto",
|
|
)
|
|
|
|
assert model == "gpt-5"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_gpt_5_tools_without_summary_stays_chat():
|
|
"""gpt-5 with tools + reasoning_effort but no summary should stay on chat."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 128000}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-5",
|
|
custom_llm_provider="openai",
|
|
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
|
reasoning_effort="medium",
|
|
reasoning_summary=None,
|
|
)
|
|
|
|
assert model == "gpt-5"
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
@patch("litellm.completion_extras.responses_api_bridge.completion")
|
|
def test_gpt_5_4_responses_bridge_preserves_reasoning_summary_dict(
|
|
mock_responses_completion,
|
|
):
|
|
"""When routed to Responses, preserve reasoning_effort summary dict."""
|
|
mock_responses_completion.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
litellm.completion(
|
|
model="gpt-5.4",
|
|
messages=[{"role": "user", "content": "What is the capital of France?"}],
|
|
tools=[
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "get_capital",
|
|
"description": "Get the capital of a country",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"country": {"type": "string"}},
|
|
},
|
|
},
|
|
}
|
|
],
|
|
reasoning_effort={"effort": "xhigh", "summary": "detailed"},
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert mock_responses_completion.called is True
|
|
optional_params = mock_responses_completion.call_args.kwargs["optional_params"]
|
|
assert optional_params["reasoning_effort"] == {
|
|
"effort": "xhigh",
|
|
"summary": "detailed",
|
|
}
|
|
|
|
|
|
@pytest.mark.parametrize("reasoning_effort", ["high", {"effort": "high"}])
|
|
def test_responses_bridge_preserves_reasoning_effort_with_drop_params(
|
|
reasoning_effort,
|
|
restore_model_registry,
|
|
respx_mock: respx.MockRouter,
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
):
|
|
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
|
response_body: Final = {
|
|
"id": "resp_test",
|
|
"object": "response",
|
|
"created_at": 1734366691,
|
|
"status": "completed",
|
|
"model": "test-responses-bridge",
|
|
"output": [
|
|
{
|
|
"type": "message",
|
|
"id": "msg_1",
|
|
"status": "completed",
|
|
"role": "assistant",
|
|
"content": [{"type": "output_text", "text": "Done.", "annotations": []}],
|
|
}
|
|
],
|
|
"parallel_tool_calls": True,
|
|
"usage": {
|
|
"input_tokens": 1,
|
|
"output_tokens": 1,
|
|
"total_tokens": 2,
|
|
"output_tokens_details": {"reasoning_tokens": 0},
|
|
},
|
|
"error": None,
|
|
"incomplete_details": None,
|
|
"instructions": None,
|
|
"metadata": None,
|
|
"temperature": None,
|
|
"tool_choice": "auto",
|
|
"tools": [],
|
|
"top_p": None,
|
|
"max_output_tokens": None,
|
|
"previous_response_id": None,
|
|
"reasoning": None,
|
|
"truncation": None,
|
|
"user": None,
|
|
}
|
|
response_route: Final = respx_mock.post("https://api.perplexity.ai/v1/responses").respond(json=response_body)
|
|
model: Final = "perplexity/test-responses-bridge"
|
|
litellm.register_model(
|
|
{
|
|
model: {
|
|
"litellm_provider": "perplexity",
|
|
"mode": "responses",
|
|
"supports_reasoning": False,
|
|
"input_cost_per_token": 0.0,
|
|
"output_cost_per_token": 0.0,
|
|
}
|
|
},
|
|
persist_across_reloads=False,
|
|
)
|
|
|
|
litellm.completion(
|
|
model=model,
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
reasoning_effort=reasoning_effort,
|
|
drop_params=True,
|
|
api_key="fake-key",
|
|
api_base="https://api.perplexity.ai",
|
|
)
|
|
|
|
request_body: Final = json.loads(response_route.calls[0].request.content)
|
|
assert request_body["reasoning"] == {"effort": "high"}
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"model, model_info, expected_model_param, expected_base_model_param",
|
|
[
|
|
("gemini/gemini-3.1-pro", None, "gemini-3.1-pro", None),
|
|
(
|
|
"gemini/gemini-3.1-pro",
|
|
{"base_model": "gemini-3.1-pro-preview"},
|
|
"gemini-3.1-pro",
|
|
"gemini-3.1-pro-preview",
|
|
),
|
|
],
|
|
)
|
|
def test_completion_optional_params_base_model(
|
|
model: str,
|
|
model_info: dict | None,
|
|
expected_model_param: str,
|
|
expected_base_model_param: str | None,
|
|
):
|
|
"""``model_info.base_model`` must reach ``get_optional_params`` as ``base_model``
|
|
(an additive capability hint), without overwriting ``model`` with the label.
|
|
|
|
Regression for #29618: overwriting ``model`` with a friendly ``base_model``
|
|
label made Bedrock drop ``tools``/``tool_choice`` under ``drop_params``."""
|
|
with patch("litellm.main.get_optional_params") as mock_get_optional_params:
|
|
mock_get_optional_params.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
kwargs = {
|
|
"model": model,
|
|
"messages": [{"role": "user", "content": "What is the capital of France?"}],
|
|
"api_key": "fake-key",
|
|
"mock_response": "Hey, how's it going?",
|
|
}
|
|
if model_info is not None:
|
|
kwargs["model_info"] = model_info
|
|
|
|
litellm.completion(**kwargs)
|
|
|
|
assert mock_get_optional_params.called is True
|
|
call_kwargs = mock_get_optional_params.call_args.kwargs
|
|
assert call_kwargs["model"] == expected_model_param
|
|
assert call_kwargs["base_model"] == expected_base_model_param
|
|
|
|
|
|
@patch("litellm.completion_extras.responses_api_bridge.completion")
|
|
def test_gpt_5_4_responses_bridge_merges_reasoning_summary_kwarg_without_tools(
|
|
mock_responses_completion,
|
|
):
|
|
"""reasoningSummary without tools should route and merge into reasoning_effort dict."""
|
|
mock_responses_completion.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
litellm.completion(
|
|
model="gpt-5.4",
|
|
messages=[{"role": "user", "content": "ok"}],
|
|
reasoning_effort="medium",
|
|
reasoningSummary="auto",
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert mock_responses_completion.called is True
|
|
optional_params = mock_responses_completion.call_args.kwargs["optional_params"]
|
|
assert optional_params["reasoning_effort"] == {
|
|
"effort": "medium",
|
|
"summary": "auto",
|
|
}
|
|
assert "reasoningSummary" not in optional_params
|
|
assert "reasoning_summary" not in optional_params
|
|
|
|
|
|
@patch("litellm.completion_extras.responses_api_bridge.completion")
|
|
def test_responses_bridge_preserves_reasoning_summary_without_effort(
|
|
mock_responses_completion,
|
|
):
|
|
"""Reasoning summary should survive responses routing even without effort."""
|
|
mock_responses_completion.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
with patch.object(litellm, "route_all_chat_openai_to_responses", True):
|
|
litellm.completion(
|
|
model="gpt-4o",
|
|
messages=[{"role": "user", "content": "ok"}],
|
|
reasoningSummary="auto",
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert mock_responses_completion.called is True
|
|
optional_params = mock_responses_completion.call_args.kwargs["optional_params"]
|
|
assert optional_params["reasoning_effort"] == {"summary": "auto"}
|
|
assert "reasoningSummary" not in optional_params
|
|
assert "reasoning_summary" not in optional_params
|
|
|
|
|
|
@patch("litellm.completion_extras.responses_api_bridge.completion")
|
|
def test_gpt_5_responses_bridge_tools_and_reasoning_summary(
|
|
mock_responses_completion,
|
|
):
|
|
"""Bare gpt-5 with tools + reasoningSummary should bridge (OpenCode-style)."""
|
|
mock_responses_completion.return_value = MagicMock()
|
|
|
|
import litellm
|
|
|
|
litellm.completion(
|
|
model="gpt-5",
|
|
messages=[{"role": "user", "content": "ok"}],
|
|
tools=[
|
|
{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "apply_patch",
|
|
"parameters": {"type": "object", "properties": {}},
|
|
},
|
|
}
|
|
],
|
|
tool_choice="auto",
|
|
reasoning_effort="medium",
|
|
reasoningSummary="auto",
|
|
stream=True,
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert mock_responses_completion.called is True
|
|
optional_params = mock_responses_completion.call_args.kwargs["optional_params"]
|
|
assert optional_params.get("reasoning_effort") == {
|
|
"effort": "medium",
|
|
"summary": "auto",
|
|
}
|
|
|
|
|
|
def test_responses_api_bridge_check_handles_exception():
|
|
"""Test that responses_api_bridge_check handles exceptions and still processes responses/ models."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.side_effect = Exception("Model not found")
|
|
|
|
model_info, model = responses_api_bridge_check(
|
|
model="responses/custom-model", custom_llm_provider="custom"
|
|
)
|
|
|
|
assert model == "custom-model"
|
|
assert model_info["mode"] == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_global_flag_routes_openai():
|
|
"""When route_all_chat_openai_to_responses is True, any OpenAI model routes to responses."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch.object(litellm, "route_all_chat_openai_to_responses", True):
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-4o",
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model == "gpt-4o"
|
|
assert model_info.get("mode") == "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_global_flag_does_not_affect_azure():
|
|
"""route_all_chat_openai_to_responses should not affect Azure models."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch.object(litellm, "route_all_chat_openai_to_responses", True):
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 4096}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-4o",
|
|
custom_llm_provider="azure",
|
|
)
|
|
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
def test_responses_api_bridge_check_global_flag_default_false():
|
|
"""By default, route_all_chat_openai_to_responses is False and doesn't affect routing."""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
with patch.object(litellm, "route_all_chat_openai_to_responses", False):
|
|
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
|
mock_get_model_info.return_value = {"max_tokens": 4096}
|
|
model_info, model = responses_api_bridge_check(
|
|
model="gpt-4o",
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model_info.get("mode") != "responses"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_async_mock_delay():
|
|
"""Use asyncio await for mock delay on acompletion"""
|
|
import time
|
|
|
|
from litellm import acompletion
|
|
|
|
start_time = time.time()
|
|
result = await acompletion(
|
|
model="gpt-3.5-turbo",
|
|
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
|
mock_delay=0.01,
|
|
mock_response="Hello world",
|
|
)
|
|
end_time = time.time()
|
|
delay = end_time - start_time
|
|
assert delay >= 0.01
|
|
|
|
|
|
def test_stream_chunk_builder_keeps_tool_calls_carried_only_by_a_later_choice_of_a_multi_choice_chunk():
|
|
from litellm import stream_chunk_builder
|
|
from litellm.types.utils import (
|
|
ChatCompletionDeltaToolCall,
|
|
Delta,
|
|
Function,
|
|
ModelResponseStream,
|
|
StreamingChoices,
|
|
)
|
|
|
|
def chunk(choices: list[StreamingChoices]) -> ModelResponseStream:
|
|
return ModelResponseStream(
|
|
id="chatcmpl-multi-choice",
|
|
created=1751934860,
|
|
model="gpt-4.1-mini",
|
|
object="chat.completion.chunk",
|
|
choices=choices,
|
|
)
|
|
|
|
chunks = [
|
|
chunk(
|
|
[
|
|
StreamingChoices(index=0, delta=Delta(role="assistant", content="hello")),
|
|
StreamingChoices(
|
|
index=1,
|
|
delta=Delta(
|
|
role="assistant",
|
|
tool_calls=[
|
|
ChatCompletionDeltaToolCall(
|
|
id="call_1",
|
|
index=0,
|
|
type="function",
|
|
function=Function(name="lookup_fruit", arguments='{"fruit":'),
|
|
)
|
|
],
|
|
),
|
|
),
|
|
]
|
|
),
|
|
chunk(
|
|
[
|
|
StreamingChoices(index=0, delta=Delta(content=" world"), finish_reason="stop"),
|
|
StreamingChoices(
|
|
index=1,
|
|
delta=Delta(
|
|
tool_calls=[ChatCompletionDeltaToolCall(index=0, function=Function(arguments='"kiwi"}'))]
|
|
),
|
|
finish_reason="tool_calls",
|
|
),
|
|
]
|
|
),
|
|
]
|
|
|
|
response = stream_chunk_builder(chunks=chunks)
|
|
|
|
tool_calls = response.choices[0].message.tool_calls
|
|
assert tool_calls is not None
|
|
assert [(call.id, call.function.name, call.function.arguments) for call in tool_calls] == [
|
|
("call_1", "lookup_fruit", '{"fruit":"kiwi"}')
|
|
]
|
|
|
|
|
|
def test_stream_chunk_builder_thinking_blocks():
|
|
from litellm import stream_chunk_builder
|
|
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
|
|
|
|
chunks = [
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content="I need to summar",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "I need to summar",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "I need to summar",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role="assistant",
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content="ize the previous agent's thinking process into a",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "ize the previous agent's thinking process into a",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "ize the previous agent's thinking process into a",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content=" short description. Based on the input data provide",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " short description. Based on the input data provide",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " short description. Based on the input data provide",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content="d, it seems the agent was planning to refine their search",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "d, it seems the agent was planning to refine their search",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "d, it seems the agent was planning to refine their search",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content=" to focus more on technical aspects of home automation and home",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " to focus more on technical aspects of home automation and home",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " to focus more on technical aspects of home automation and home",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content=" energy system management.\n\nI'll create a brief",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " energy system management.\n\nI'll create a brief",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " energy system management.\n\nI'll create a brief",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content=" summary of what the agent was doing.",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " summary of what the agent was doing.",
|
|
"signature": None,
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": " summary of what the agent was doing.",
|
|
"signature": None,
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=0,
|
|
delta=Delta(
|
|
reasoning_content="",
|
|
thinking_blocks=[
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "",
|
|
"signature": "ErUBCkYIBRgCIkAKBSMkB2+MBF643wiWxlERsGXVdlhbPx9lnTIbygzjFIeZ5uhTV+HNWDon9vQV4hmXvAKwQfwS8vkNFB366l05Egzt2U18IpRrZRyQn1UaDDdYvKHYP8Ps1IbWjSIw8eSYOU9gtqNcwR6D0wY7iOPx2GliDEatLI5rSs96CByoTIoADL2M5bX8KP0jEpbHKh0ccYryigdH/3J8EiFt/BmGUceVASP5l9r22dFWiBgC",
|
|
}
|
|
],
|
|
provider_specific_fields={
|
|
"thinking_blocks": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "",
|
|
"signature": "ErUBCkYIBRgCIkAKBSMkB2+MBF643wiWxlERsGXVdlhbPx9lnTIbygzjFIeZ5uhTV+HNWDon9vQV4hmXvAKwQfwS8vkNFB366l05Egzt2U18IpRrZRyQn1UaDDdYvKHYP8Ps1IbWjSIw8eSYOU9gtqNcwR6D0wY7iOPx2GliDEatLI5rSs96CByoTIoADL2M5bX8KP0jEpbHKh0ccYryigdH/3J8EiFt/BmGUceVASP5l9r22dFWiBgC",
|
|
}
|
|
]
|
|
},
|
|
content="",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content='{"a',
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content='gent_doing"',
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content=': "Re',
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content="searching",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content=" technic",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content="al aspect",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content="s of home au",
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason=None,
|
|
index=1,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content='tomation"}',
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
citations=None,
|
|
),
|
|
ModelResponseStream(
|
|
id="chatcmpl-e8febeb7-cf7d-4947-9417-59ae5e6989f9",
|
|
created=1751934860,
|
|
model="claude-3-7-sonnet-latest",
|
|
object="chat.completion.chunk",
|
|
system_fingerprint=None,
|
|
choices=[
|
|
StreamingChoices(
|
|
finish_reason="tool_calls",
|
|
index=0,
|
|
delta=Delta(
|
|
provider_specific_fields=None,
|
|
content=None,
|
|
role=None,
|
|
function_call=None,
|
|
tool_calls=None,
|
|
audio=None,
|
|
),
|
|
logprobs=None,
|
|
)
|
|
],
|
|
provider_specific_fields=None,
|
|
),
|
|
]
|
|
|
|
response = stream_chunk_builder(chunks=chunks)
|
|
print(response)
|
|
|
|
assert response is not None
|
|
assert response.choices[0].message.content is not None
|
|
assert response.choices[0].message.thinking_blocks is not None
|
|
|
|
|
|
from litellm.llms.openai.openai import OpenAIChatCompletion
|
|
|
|
|
|
def throw_retryable_error(*_, **__):
|
|
raise RuntimeError("BOOM")
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_retrying() -> None:
|
|
litellm.num_retries = 10
|
|
with (
|
|
patch.object(
|
|
OpenAIChatCompletion,
|
|
"make_openai_chat_completion_request",
|
|
side_effect=throw_retryable_error,
|
|
) as mock_request,
|
|
pytest.raises(litellm.InternalServerError, match="LiteLLM Retried: 10 times"),
|
|
):
|
|
await litellm.acompletion(
|
|
model="gpt-4o-mini",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
)
|
|
|
|
|
|
def test_anthropic_disable_url_suffix_env_var():
|
|
"""Test that LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX prevents /v1/messages suffix."""
|
|
import os
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
from litellm import completion
|
|
|
|
# Test with environment variable disabled (default behavior)
|
|
with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}):
|
|
actual_api_base = None
|
|
|
|
with patch("litellm.main.anthropic_chat_completions") as mock_anthropic:
|
|
|
|
def capture_completion(**kwargs):
|
|
nonlocal actual_api_base
|
|
actual_api_base = kwargs.get("api_base")
|
|
mock_response = MagicMock()
|
|
mock_response.choices = [MagicMock()]
|
|
return mock_response
|
|
|
|
mock_anthropic.completion = capture_completion
|
|
|
|
# This should append /v1/messages
|
|
completion(
|
|
model="anthropic/claude-3-sonnet",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
api_key="test-key",
|
|
)
|
|
|
|
# Verify the api_base has /v1/messages appended
|
|
assert actual_api_base.endswith("/v1/messages")
|
|
assert actual_api_base == "https://api.example.com/v1/messages"
|
|
|
|
# Test with environment variable enabled
|
|
with patch.dict(
|
|
os.environ,
|
|
{
|
|
"ANTHROPIC_API_BASE": "https://api.example.com/custom/path",
|
|
"LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX": "true",
|
|
},
|
|
):
|
|
actual_api_base = None
|
|
|
|
with patch("litellm.main.anthropic_chat_completions") as mock_anthropic:
|
|
|
|
def capture_completion(**kwargs):
|
|
nonlocal actual_api_base
|
|
actual_api_base = kwargs.get("api_base")
|
|
mock_response = MagicMock()
|
|
mock_response.choices = [MagicMock()]
|
|
return mock_response
|
|
|
|
mock_anthropic.completion = capture_completion
|
|
|
|
# This should NOT append /v1/messages
|
|
completion(
|
|
model="anthropic/claude-3-sonnet",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
api_key="test-key",
|
|
)
|
|
|
|
# Verify the api_base does not have /v1/messages appended
|
|
assert actual_api_base == "https://api.example.com/custom/path"
|
|
assert not actual_api_base.endswith("/v1/messages")
|
|
|
|
|
|
def test_anthropic_text_disable_url_suffix_env_var():
|
|
"""Test that LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX prevents /v1/complete suffix for anthropic_text."""
|
|
import os
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
from litellm import completion
|
|
|
|
# Test with environment variable disabled (default behavior)
|
|
with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}):
|
|
actual_api_base = None
|
|
|
|
with patch("litellm.main.base_llm_http_handler") as mock_handler:
|
|
|
|
def capture_completion(**kwargs):
|
|
nonlocal actual_api_base
|
|
actual_api_base = kwargs.get("api_base")
|
|
return MagicMock()
|
|
|
|
mock_handler.completion = capture_completion
|
|
|
|
# This should append /v1/complete
|
|
completion(
|
|
model="anthropic_text/claude-instant-1",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
api_key="test-key",
|
|
)
|
|
|
|
# Verify the api_base has /v1/complete appended
|
|
assert actual_api_base.endswith("/v1/complete")
|
|
assert actual_api_base == "https://api.example.com/v1/complete"
|
|
|
|
# Test with environment variable enabled
|
|
with patch.dict(
|
|
os.environ,
|
|
{
|
|
"ANTHROPIC_API_BASE": "https://api.example.com/custom/complete",
|
|
"LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX": "true",
|
|
},
|
|
):
|
|
actual_api_base = None
|
|
|
|
with patch("litellm.main.base_llm_http_handler") as mock_handler:
|
|
|
|
def capture_completion(**kwargs):
|
|
nonlocal actual_api_base
|
|
actual_api_base = kwargs.get("api_base")
|
|
return MagicMock()
|
|
|
|
mock_handler.completion = capture_completion
|
|
|
|
# This should NOT append /v1/complete
|
|
completion(
|
|
model="anthropic_text/claude-instant-1",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
api_key="test-key",
|
|
)
|
|
|
|
# Verify the api_base does not have /v1/complete appended
|
|
assert actual_api_base == "https://api.example.com/custom/complete"
|
|
assert not actual_api_base.endswith("/v1/complete")
|
|
|
|
|
|
def test_image_edit_merges_headers_and_extra_headers():
|
|
from litellm.images.main import base_llm_http_handler
|
|
|
|
combined_headers = {
|
|
"x-test-header-one": "value-1",
|
|
"x-test-header-two": "value-2",
|
|
}
|
|
|
|
mock_image_edit_config = MagicMock()
|
|
mock_image_edit_config.get_supported_openai_params.return_value = set()
|
|
mock_image_edit_config.map_openai_params.side_effect = lambda **kwargs: dict(
|
|
kwargs["image_edit_optional_params"]
|
|
)
|
|
|
|
with (
|
|
patch(
|
|
"litellm.images.main.ProviderConfigManager.get_provider_image_edit_config",
|
|
return_value=mock_image_edit_config,
|
|
) as mock_config,
|
|
patch.object(
|
|
base_llm_http_handler,
|
|
"image_edit_handler",
|
|
return_value="ok",
|
|
) as mock_handler,
|
|
):
|
|
response = litellm.image_edit(
|
|
image=MagicMock(name="image"),
|
|
prompt="test",
|
|
model="azure/gpt-image-1",
|
|
headers={"x-test-header-one": "value-1"},
|
|
extra_headers={
|
|
"x-test-header-two": "value-2",
|
|
},
|
|
)
|
|
|
|
assert response == "ok"
|
|
mock_config.assert_called_once()
|
|
|
|
handler_kwargs = mock_handler.call_args.kwargs
|
|
assert handler_kwargs["extra_headers"] == combined_headers
|
|
assert "extra_headers" not in handler_kwargs["image_edit_optional_request_params"]
|
|
|
|
|
|
@pytest.mark.parametrize("metadata_key", ("metadata", "litellm_metadata"))
|
|
@pytest.mark.parametrize("input_tokens", (51234, 0))
|
|
def test_mock_completion_usage_reports_admission_input_tokens(metadata_key: str, input_tokens: int):
|
|
response = litellm.completion(
|
|
model="anthropic/claude-sonnet-5",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
**{metadata_key: {"user_api_key_budget_reservation": {"reserved_cost": 1.0, "input_tokens": input_tokens}}},
|
|
)
|
|
|
|
assert response.usage.prompt_tokens == input_tokens
|
|
assert response.usage.total_tokens == input_tokens + response.usage.completion_tokens
|
|
|
|
|
|
def test_mock_completion_usage_falls_back_to_default_without_admission_count():
|
|
response = litellm.completion(
|
|
model="anthropic/claude-sonnet-5",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
metadata={"user_api_key_budget_reservation": {"reserved_cost": 1.0}},
|
|
)
|
|
|
|
assert response.usage.prompt_tokens == litellm_main.DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT
|
|
|
|
|
|
_ADMISSION_INPUT_TOKENS: Final = 51234
|
|
|
|
|
|
def _admission_metadata(input_tokens: int) -> dict[str, object]: # mutable-ok: logging writes into metadata
|
|
return {"user_api_key_budget_reservation": {"reserved_cost": 1.0, "input_tokens": input_tokens}}
|
|
|
|
|
|
_ADMISSION_METADATA: Final = _admission_metadata(_ADMISSION_INPUT_TOKENS)
|
|
_MOCK_STREAM_MESSAGES: Final = [{"role": "user", "content": "hello " * 200}]
|
|
_STREAM_CHUNK_BUILDER_TOKEN_COUNTER: Final = "litellm.litellm_core_utils.streaming_chunk_builder_utils.token_counter"
|
|
|
|
|
|
def _prompt_token_counter_calls(token_counter: MagicMock) -> list[object]:
|
|
return [call for call in token_counter.call_args_list if call.kwargs.get("messages") is not None]
|
|
|
|
|
|
def _client_usage_chunks(chunks: list[ModelResponseStream]) -> list[Usage]:
|
|
return [chunk.usage for chunk in chunks if getattr(chunk, "usage", None) is not None]
|
|
|
|
|
|
@pytest.mark.parametrize("n", (None, 2))
|
|
def test_mock_completion_stream_usage_reports_admission_input_tokens_without_tokenizer_fallback(n: int | None):
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
chunks: Final = list(
|
|
litellm.completion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
n=n,
|
|
stream_options={"include_usage": True},
|
|
metadata=_ADMISSION_METADATA,
|
|
)
|
|
)
|
|
|
|
usage_chunks: Final = _client_usage_chunks(chunks)
|
|
assert len(usage_chunks) == 1
|
|
assert usage_chunks[0].prompt_tokens == _ADMISSION_INPUT_TOKENS
|
|
assert usage_chunks[0].completion_tokens == litellm_main.DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT
|
|
assert usage_chunks[0].total_tokens == _ADMISSION_INPUT_TOKENS + usage_chunks[0].completion_tokens
|
|
assert _prompt_token_counter_calls(token_counter) == []
|
|
assert all(chunk.choices for chunk in chunks[:-1])
|
|
assert {chunk.id for chunk in chunks} == {chunks[0].id}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize("n", (None, 2))
|
|
async def test_mock_acompletion_stream_usage_reports_admission_input_tokens_without_tokenizer_fallback(
|
|
n: int | None,
|
|
):
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
response: Final = await litellm.acompletion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
n=n,
|
|
stream_options={"include_usage": True},
|
|
litellm_metadata=_ADMISSION_METADATA,
|
|
)
|
|
chunks: Final = [chunk async for chunk in response]
|
|
|
|
usage_chunks: Final = _client_usage_chunks(chunks)
|
|
assert len(usage_chunks) == 1
|
|
assert usage_chunks[0].prompt_tokens == _ADMISSION_INPUT_TOKENS
|
|
assert usage_chunks[0].total_tokens == _ADMISSION_INPUT_TOKENS + usage_chunks[0].completion_tokens
|
|
assert _prompt_token_counter_calls(token_counter) == []
|
|
assert all(chunk.choices for chunk in chunks[:-1])
|
|
assert {chunk.id for chunk in chunks} == {chunks[0].id}
|
|
|
|
|
|
def test_mock_completion_stream_without_include_usage_hides_usage_chunk_but_logs_admission_count():
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
chunks: Final = list(
|
|
litellm.completion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
metadata=_ADMISSION_METADATA,
|
|
)
|
|
)
|
|
|
|
assert _client_usage_chunks(chunks) == []
|
|
assert all(len(chunk.choices) == 1 for chunk in chunks)
|
|
assert chunks[-1]._hidden_params["usage"].prompt_tokens == _ADMISSION_INPUT_TOKENS
|
|
assert _prompt_token_counter_calls(token_counter) == []
|
|
|
|
|
|
def test_mock_completion_stream_with_empty_stream_options_completes_and_logs_admission_count():
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
chunks: Final = list(
|
|
litellm.completion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
stream_options={},
|
|
metadata=_ADMISSION_METADATA,
|
|
)
|
|
)
|
|
|
|
assert "".join(chunk.choices[0].delta.content or "" for chunk in chunks) == "ok"
|
|
assert _client_usage_chunks(chunks) == []
|
|
assert _prompt_token_counter_calls(token_counter) == []
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_mock_acompletion_stream_with_empty_stream_options_completes_and_logs_admission_count():
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
response: Final = await litellm.acompletion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
stream_options={},
|
|
litellm_metadata=_ADMISSION_METADATA,
|
|
)
|
|
chunks: Final = [chunk async for chunk in response]
|
|
|
|
assert "".join(chunk.choices[0].delta.content or "" for chunk in chunks) == "ok"
|
|
assert _client_usage_chunks(chunks) == []
|
|
assert _prompt_token_counter_calls(token_counter) == []
|
|
|
|
|
|
def test_mock_completion_stream_without_admission_count_falls_back_to_tokenizer():
|
|
expected_prompt_tokens: Final = litellm.token_counter(model="openai/gpt-5.4-mini", messages=_MOCK_STREAM_MESSAGES)
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
chunks: Final = list(
|
|
litellm.completion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
stream_options={"include_usage": True},
|
|
metadata={"user_api_key_budget_reservation": {"reserved_cost": 1.0}},
|
|
)
|
|
)
|
|
|
|
usage_chunks: Final = _client_usage_chunks(chunks)
|
|
assert len(usage_chunks) == 1
|
|
assert usage_chunks[0].prompt_tokens == expected_prompt_tokens
|
|
assert usage_chunks[0].total_tokens == expected_prompt_tokens + usage_chunks[0].completion_tokens
|
|
assert len(_prompt_token_counter_calls(token_counter)) >= 1
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_mock_acompletion_stream_without_admission_count_falls_back_to_tokenizer():
|
|
expected_prompt_tokens: Final = litellm.token_counter(model="openai/gpt-5.4-mini", messages=_MOCK_STREAM_MESSAGES)
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
response: Final = await litellm.acompletion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
stream_options={"include_usage": True},
|
|
)
|
|
chunks: Final = [chunk async for chunk in response]
|
|
|
|
usage_chunks: Final = _client_usage_chunks(chunks)
|
|
assert len(usage_chunks) == 1
|
|
assert usage_chunks[0].prompt_tokens == expected_prompt_tokens
|
|
assert len(_prompt_token_counter_calls(token_counter)) >= 1
|
|
|
|
|
|
def _usage_triple(usage: Usage) -> tuple[int, int, int]:
|
|
return (usage.prompt_tokens, usage.completion_tokens, usage.total_tokens)
|
|
|
|
|
|
@pytest.mark.parametrize("input_tokens", (_ADMISSION_INPUT_TOKENS, 0))
|
|
def test_mock_completion_stream_and_non_stream_report_the_same_admission_usage(input_tokens: int):
|
|
metadata: Final = _admission_metadata(input_tokens)
|
|
non_stream: Final = litellm.completion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
metadata=metadata,
|
|
)
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
chunks: Final = list(
|
|
litellm.completion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=_MOCK_STREAM_MESSAGES,
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
stream_options={"include_usage": True},
|
|
metadata=metadata,
|
|
)
|
|
)
|
|
|
|
assert _usage_triple(non_stream.usage) == _usage_triple(_client_usage_chunks(chunks)[0])
|
|
assert non_stream.usage.prompt_tokens == input_tokens
|
|
assert _prompt_token_counter_calls(token_counter) == []
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_mock_acompletion_stream_reports_zero_admission_input_tokens_without_tokenizer_fallback():
|
|
with patch(_STREAM_CHUNK_BUILDER_TOKEN_COUNTER, wraps=litellm.token_counter) as token_counter:
|
|
response: Final = await litellm.acompletion(
|
|
model="openai/gpt-5.4-mini",
|
|
messages=[{"role": "user", "content": ""}],
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
stream_options={"include_usage": True},
|
|
litellm_metadata=_admission_metadata(0),
|
|
)
|
|
chunks: Final = [chunk async for chunk in response]
|
|
|
|
usage_chunks: Final = _client_usage_chunks(chunks)
|
|
assert len(usage_chunks) == 1
|
|
assert _usage_triple(usage_chunks[0]) == (0, usage_chunks[0].completion_tokens, usage_chunks[0].completion_tokens)
|
|
assert _prompt_token_counter_calls(token_counter) == []
|
|
|
|
|
|
def test_mock_text_completion_stream_and_non_stream_report_the_same_zero_admission_usage():
|
|
metadata: Final = _admission_metadata(0)
|
|
non_stream: Final = litellm.text_completion(
|
|
model="openai/gpt-5.4-mini", prompt="", mock_response="ok", api_key="mock", metadata=metadata
|
|
)
|
|
chunks: Final = list(
|
|
litellm.text_completion(
|
|
model="openai/gpt-5.4-mini",
|
|
prompt="",
|
|
mock_response="ok",
|
|
api_key="mock",
|
|
stream=True,
|
|
stream_options={"include_usage": True},
|
|
metadata=metadata,
|
|
)
|
|
)
|
|
|
|
stream_usages: Final = tuple(chunk.usage for chunk in chunks if getattr(chunk, "usage", None) is not None)
|
|
assert len(stream_usages) == 1
|
|
assert _usage_triple(non_stream.usage) == _usage_triple(stream_usages[0])
|
|
assert non_stream.usage.prompt_tokens == 0
|
|
|
|
|
|
def test_mock_completion_stream_with_model_response():
|
|
"""Test that mock_completion correctly handles stream=True with a ModelResponse as mock_response."""
|
|
from litellm import completion
|
|
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
|
|
|
# Create a ModelResponse object
|
|
mock_model_response = ModelResponse(
|
|
id="chatcmpl-test-123",
|
|
created=1234567890,
|
|
model="gpt-4o-mini",
|
|
object="chat.completion",
|
|
choices=[
|
|
Choices(
|
|
finish_reason="stop",
|
|
index=0,
|
|
message=Message(
|
|
content="This is a test response",
|
|
role="assistant",
|
|
),
|
|
)
|
|
],
|
|
usage=Usage(
|
|
prompt_tokens=10,
|
|
completion_tokens=20,
|
|
total_tokens=30,
|
|
),
|
|
)
|
|
|
|
# Call completion with stream=True and mock_response as ModelResponse
|
|
response = completion(
|
|
model="gpt-4o-mini",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
stream=True,
|
|
mock_response=mock_model_response,
|
|
)
|
|
|
|
# Verify that the response is a stream
|
|
assert response is not None
|
|
|
|
# Collect all chunks from the stream
|
|
chunks = []
|
|
for chunk in response:
|
|
chunks.append(chunk)
|
|
print(f"Chunk: {chunk}")
|
|
|
|
# Verify we got chunks
|
|
assert len(chunks) > 0
|
|
|
|
# Verify the content is streamed correctly
|
|
accumulated_content = ""
|
|
for chunk in chunks:
|
|
if (
|
|
hasattr(chunk.choices[0].delta, "content")
|
|
and chunk.choices[0].delta.content
|
|
):
|
|
accumulated_content += chunk.choices[0].delta.content
|
|
|
|
assert "This is a test response" in accumulated_content or len(chunks) > 0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_async_mock_completion_stream_with_model_response():
|
|
"""Test that async mock_completion correctly handles stream=True with a ModelResponse as mock_response."""
|
|
from litellm import acompletion
|
|
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
|
|
|
# Create a ModelResponse object
|
|
mock_model_response = ModelResponse(
|
|
id="chatcmpl-test-456",
|
|
created=1234567890,
|
|
model="gpt-4o-mini",
|
|
object="chat.completion",
|
|
choices=[
|
|
Choices(
|
|
finish_reason="stop",
|
|
index=0,
|
|
message=Message(
|
|
content="This is an async test response",
|
|
role="assistant",
|
|
),
|
|
)
|
|
],
|
|
usage=Usage(
|
|
prompt_tokens=15,
|
|
completion_tokens=25,
|
|
total_tokens=40,
|
|
),
|
|
)
|
|
|
|
# Call acompletion with stream=True and mock_response as ModelResponse
|
|
response = await acompletion(
|
|
model="gpt-4o-mini",
|
|
messages=[{"role": "user", "content": "Hello async"}],
|
|
stream=True,
|
|
mock_response=mock_model_response,
|
|
)
|
|
|
|
# Verify that the response is a stream
|
|
assert response is not None
|
|
|
|
# Collect all chunks from the stream
|
|
chunks = []
|
|
async for chunk in response:
|
|
chunks.append(chunk)
|
|
print(f"Async Chunk: {chunk}")
|
|
|
|
# Verify we got chunks
|
|
assert len(chunks) > 0
|
|
|
|
# Verify the content is streamed correctly
|
|
accumulated_content = ""
|
|
for chunk in chunks:
|
|
if (
|
|
hasattr(chunk.choices[0].delta, "content")
|
|
and chunk.choices[0].delta.content
|
|
):
|
|
accumulated_content += chunk.choices[0].delta.content
|
|
|
|
assert "This is an async test response" in accumulated_content or len(chunks) > 0
|
|
|
|
|
|
class TestCallTypesOCR:
|
|
"""Test that OCR call types are properly defined in CallTypes enum.
|
|
|
|
Fixes https://github.com/BerriAI/litellm/issues/17381
|
|
"""
|
|
|
|
def test_ocr_call_type_exists(self):
|
|
"""Test that CallTypes.ocr exists and has correct value."""
|
|
from litellm.types.utils import CallTypes
|
|
|
|
assert hasattr(CallTypes, "ocr")
|
|
assert CallTypes.ocr.value == "ocr"
|
|
|
|
def test_aocr_call_type_exists(self):
|
|
"""Test that CallTypes.aocr exists and has correct value."""
|
|
from litellm.types.utils import CallTypes
|
|
|
|
assert hasattr(CallTypes, "aocr")
|
|
assert CallTypes.aocr.value == "aocr"
|
|
|
|
def test_ocr_call_type_from_string(self):
|
|
"""Test that CallTypes can be constructed from 'ocr' string."""
|
|
from litellm.types.utils import CallTypes
|
|
|
|
call_type = CallTypes("ocr")
|
|
assert call_type == CallTypes.ocr
|
|
|
|
def test_aocr_call_type_from_string(self):
|
|
"""Test that CallTypes can be constructed from 'aocr' string.
|
|
|
|
This is the actual use case that was failing - the OCR endpoint
|
|
uses route_type='aocr' and guardrails try to instantiate
|
|
CallTypes('aocr').
|
|
"""
|
|
from litellm.types.utils import CallTypes
|
|
|
|
call_type = CallTypes("aocr")
|
|
assert call_type == CallTypes.aocr
|
|
|
|
|
|
def test_stream_chunk_builder_text_completion_combines_text_and_usage():
|
|
from litellm.main import stream_chunk_builder_text_completion
|
|
from litellm.types.utils import TextCompletionResponse
|
|
|
|
chunks = [
|
|
TextCompletionResponse(
|
|
id="cmpl-1",
|
|
object="text_completion",
|
|
created=1,
|
|
model="gpt-3.5-turbo-instruct",
|
|
choices=[{"text": "Hello", "index": 0, "logprobs": None, "finish_reason": None}],
|
|
),
|
|
TextCompletionResponse(
|
|
id="cmpl-1",
|
|
object="text_completion",
|
|
created=1,
|
|
model="gpt-3.5-turbo-instruct",
|
|
choices=[{"text": " world", "index": 0, "logprobs": None, "finish_reason": "stop"}],
|
|
),
|
|
]
|
|
|
|
response = stream_chunk_builder_text_completion(
|
|
chunks=chunks, messages=[{"role": "user", "content": "say hello"}]
|
|
)
|
|
|
|
assert response.choices[0].text == "Hello world"
|
|
assert response.choices[0].finish_reason == "stop"
|
|
assert response.usage.prompt_tokens > 0
|
|
assert response.usage.completion_tokens > 0
|
|
assert response.usage.total_tokens == response.usage.prompt_tokens + response.usage.completion_tokens
|
|
|
|
|
|
def test_completion_forwards_store_and_prompt_cache_key_to_openai():
|
|
"""
|
|
Regression test for https://github.com/BerriAI/litellm/issues/33184
|
|
|
|
store and prompt_cache_key are documented OpenAI chat completion params that
|
|
were accepted as supported but silently dropped before the provider request
|
|
was built, because they were not named parameters of completion() and
|
|
get_optional_params() the way safety_identifier is.
|
|
"""
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key")
|
|
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_client:
|
|
try:
|
|
litellm.completion(
|
|
model="openai/gpt-4o",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
store=False,
|
|
prompt_cache_key="test-cache-key",
|
|
client=client,
|
|
)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_client.assert_called_once()
|
|
request_body = mock_client.call_args.kwargs
|
|
assert request_body["store"] is False
|
|
assert request_body["prompt_cache_key"] == "test-cache-key"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_acompletion_forwards_store_and_prompt_cache_key_to_openai():
|
|
"""
|
|
Async variant of the store/prompt_cache_key forwarding regression test for
|
|
https://github.com/BerriAI/litellm/issues/33184
|
|
"""
|
|
from openai import AsyncOpenAI
|
|
|
|
client = AsyncOpenAI(api_key="fake-api-key")
|
|
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_client:
|
|
try:
|
|
await litellm.acompletion(
|
|
model="openai/gpt-4o",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
store=False,
|
|
prompt_cache_key="test-cache-key",
|
|
client=client,
|
|
)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_client.assert_called_once()
|
|
request_body = mock_client.call_args.kwargs
|
|
assert request_body["store"] is False
|
|
assert request_body["prompt_cache_key"] == "test-cache-key"
|
|
|
|
|
|
def test_completion_omits_store_and_prompt_cache_key_when_not_passed():
|
|
"""
|
|
When store and prompt_cache_key are not passed, they must not appear in the
|
|
outbound request body (guards against always forwarding None defaults).
|
|
"""
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key")
|
|
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_client:
|
|
try:
|
|
litellm.completion(
|
|
model="openai/gpt-4o",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
client=client,
|
|
)
|
|
except Exception as e:
|
|
print(e)
|
|
|
|
mock_client.assert_called_once()
|
|
request_body = mock_client.call_args.kwargs
|
|
assert "store" not in request_body
|
|
assert "prompt_cache_key" not in request_body
|
|
|
|
|
|
def test_completion_forwards_store_and_prompt_cache_key_to_mcp_gateway():
|
|
"""
|
|
Regression test for the MCP gateway early-return in completion(): store and
|
|
prompt_cache_key are named params, so they no longer travel via **kwargs and
|
|
must be forwarded explicitly like safety_identifier and service_tier.
|
|
"""
|
|
with patch.object(
|
|
import_module("litellm.responses.mcp.chat_completions_handler"), "acompletion_with_mcp"
|
|
) as mock_mcp:
|
|
result = litellm.completion(
|
|
model="openai/gpt-4o",
|
|
messages=[{"role": "user", "content": "Hello"}],
|
|
tools=[{"type": "mcp", "server_url": "litellm_proxy"}],
|
|
store=False,
|
|
prompt_cache_key="test-cache-key",
|
|
)
|
|
|
|
result.close()
|
|
mock_mcp.assert_called_once()
|
|
call_kwargs = mock_mcp.call_args.kwargs
|
|
assert call_kwargs["store"] is False
|
|
assert call_kwargs["prompt_cache_key"] == "test-cache-key"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(
|
|
"aws_credential_kwargs",
|
|
[
|
|
{
|
|
"aws_session_name": "litellm-gcp",
|
|
"aws_role_name": "arn:aws:iam::123456789012:role/litellm-bedrock-role",
|
|
"aws_web_identity_token": "oidc/google/108963886734710037768",
|
|
},
|
|
{
|
|
"aws_access_key_id": "AKIASTATICKEYFORTEST",
|
|
"aws_secret_access_key": "static-secret-key",
|
|
"aws_session_token": "static-session-token",
|
|
},
|
|
],
|
|
ids=["web_identity", "static_keys"],
|
|
)
|
|
async def test_acompletion_forwards_aws_credentials_through_responses_bridge(
|
|
respx_mock: respx.MockRouter, monkeypatch, aws_credential_kwargs: dict
|
|
):
|
|
from botocore.credentials import Credentials
|
|
|
|
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
|
|
|
|
original_disable_aiohttp = litellm.disable_aiohttp_transport
|
|
try:
|
|
litellm.disable_aiohttp_transport = True
|
|
monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True")
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False)
|
|
monkeypatch.delenv("BEDROCK_MANTLE_API_KEY", raising=False)
|
|
|
|
get_credentials_mock = MagicMock(return_value=Credentials("fake-key", "fake-secret"))
|
|
monkeypatch.setattr(BaseAWSLLM, "get_credentials", get_credentials_mock)
|
|
|
|
respx_mock.post("https://bedrock-mantle.us-east-2.api.aws/openai/v1/responses").respond(
|
|
json={
|
|
"id": "resp_123",
|
|
"object": "response",
|
|
"created_at": 1760144904,
|
|
"status": "completed",
|
|
"model": "openai.gpt-5.4",
|
|
"output": [
|
|
{
|
|
"type": "message",
|
|
"id": "msg_1",
|
|
"role": "assistant",
|
|
"status": "completed",
|
|
"content": [{"type": "output_text", "text": "ok", "annotations": []}],
|
|
}
|
|
],
|
|
}
|
|
)
|
|
|
|
response = await litellm.acompletion(
|
|
model="bedrock_mantle/openai.gpt-5.4",
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
api_base="https://bedrock-mantle.us-east-2.api.aws/v1",
|
|
aws_region_name="us-east-2",
|
|
num_retries=0,
|
|
**aws_credential_kwargs,
|
|
)
|
|
|
|
assert response.choices[0].message.content == "ok"
|
|
credential_kwargs = get_credentials_mock.call_args.kwargs
|
|
assert credential_kwargs["aws_region_name"] == "us-east-2"
|
|
for key, value in aws_credential_kwargs.items():
|
|
assert credential_kwargs[key] == value
|
|
authorization = respx_mock.calls.last.request.headers["Authorization"]
|
|
assert authorization.startswith("AWS4-HMAC-SHA256")
|
|
assert "fake-key" in authorization
|
|
finally:
|
|
litellm.disable_aiohttp_transport = original_disable_aiohttp
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
|
|
_GEMINI_RESPONSE_BODY = {
|
|
"candidates": [{"content": {"parts": [{"text": "hello"}], "role": "model"}, "finishReason": "STOP"}],
|
|
"usageMetadata": {"promptTokenCount": 2, "candidatesTokenCount": 1, "totalTokenCount": 3},
|
|
}
|
|
|
|
|
|
def _gemini_client_returning_a_reply():
|
|
"""An injected HTTP client whose post() answers like generativelanguage does."""
|
|
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
|
|
|
client = HTTPHandler()
|
|
request = httpx.Request("POST", "https://generativelanguage.googleapis.com/")
|
|
post = MagicMock(return_value=httpx.Response(200, json=_GEMINI_RESPONSE_BODY, request=request))
|
|
return client, post
|
|
|
|
|
|
@pytest.fixture
|
|
def restore_model_registry():
|
|
"""litellm.model_cost and the provider name sets are module-global.
|
|
|
|
register_model merges into the existing entry in place, hence the deep copy.
|
|
"""
|
|
model_cost = copy.deepcopy(litellm.model_cost)
|
|
openai_models = set(litellm.open_ai_chat_completion_models)
|
|
yield
|
|
litellm.model_cost.clear()
|
|
litellm.model_cost.update(model_cost)
|
|
litellm.open_ai_chat_completion_models.clear()
|
|
litellm.open_ai_chat_completion_models.update(openai_models)
|
|
|
|
|
|
def test_openai_model_name_does_not_outrank_explicit_provider():
|
|
"""`gemini/gpt-4o` goes to Google, not to litellm's OpenAI handler.
|
|
|
|
completion() checks `model in litellm.open_ai_chat_completion_models` ahead of
|
|
the gemini branch, so the call used to reach the OpenAI handler carrying
|
|
VertexGeminiConfig, whose transform_request raises NotImplementedError.
|
|
"""
|
|
assert "gpt-4o" in litellm.open_ai_chat_completion_models
|
|
client, post = _gemini_client_returning_a_reply()
|
|
|
|
with patch.object(client, "post", new=post):
|
|
response = litellm.completion(
|
|
model="gemini/gpt-4o",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
api_key="test-api-key",
|
|
client=client,
|
|
)
|
|
|
|
assert "generativelanguage.googleapis.com" in post.call_args.kwargs["url"]
|
|
assert "models/gpt-4o" in post.call_args.kwargs["url"]
|
|
assert response.choices[0].message.content == "hello"
|
|
|
|
|
|
def test_mislabelled_pricing_entry_does_not_reroute_provider(restore_model_registry):
|
|
"""register_model is the other way into the same failure.
|
|
|
|
An entry claiming litellm_provider "openai" adds its name to
|
|
open_ai_chat_completion_models, so one mislabelled price reroutes every later
|
|
call to that model in the process.
|
|
"""
|
|
litellm.register_model(
|
|
{
|
|
"gemini-2.5-pro": {
|
|
"litellm_provider": "openai",
|
|
"mode": "chat",
|
|
"input_cost_per_token": 1e-06,
|
|
"output_cost_per_token": 4e-06,
|
|
}
|
|
}
|
|
)
|
|
assert "gemini-2.5-pro" in litellm.open_ai_chat_completion_models
|
|
client, post = _gemini_client_returning_a_reply()
|
|
|
|
with patch.object(client, "post", new=post):
|
|
response = litellm.completion(
|
|
model="gemini/gemini-2.5-pro",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
api_key="test-api-key",
|
|
client=client,
|
|
)
|
|
|
|
assert "generativelanguage.googleapis.com" in post.call_args.kwargs["url"]
|
|
assert response.choices[0].message.content == "hello"
|
|
|
|
|
|
def test_openai_model_without_a_provider_still_routes_to_openai():
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-key")
|
|
raw_response = client.chat.completions.with_raw_response
|
|
with patch.object(raw_response, "create") as mock_create, contextlib.suppress(Exception):
|
|
litellm.completion(
|
|
model="gpt-4o",
|
|
messages=[{"role": "user", "content": "hello"}],
|
|
client=client,
|
|
)
|
|
|
|
mock_create.assert_called()
|
|
|
|
|
|
def _openai_chat_create_kwargs(client, **completion_kwargs):
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_client:
|
|
with contextlib.suppress(Exception):
|
|
litellm.completion(
|
|
messages=[{"role": "system", "content": "sys"}, {"role": "user", "content": "hi"}],
|
|
cache_control_injection_points=[{"location": "message", "role": "system"}],
|
|
client=client,
|
|
**completion_kwargs,
|
|
)
|
|
|
|
mock_client.assert_called_once()
|
|
return mock_client.call_args.kwargs
|
|
|
|
|
|
@pytest.fixture
|
|
def _no_openai_api_base_override(monkeypatch):
|
|
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
|
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
|
monkeypatch.setattr(litellm, "api_base", None)
|
|
|
|
|
|
@pytest.mark.usefixtures("_no_openai_api_base_override")
|
|
def test_completion_custom_api_base_sends_no_prompt_cache_breakpoint_for_gpt_5_6():
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key", base_url="http://127.0.0.1:9/v1")
|
|
request_body = _openai_chat_create_kwargs(client, model="gpt-5.6", api_base="http://127.0.0.1:9/v1")
|
|
|
|
assert request_body["messages"][0] == {"role": "system", "content": "sys", "cache_control": {"type": "ephemeral"}}
|
|
assert "prompt_cache_breakpoint" not in json.dumps(request_body["messages"])
|
|
assert "prompt_cache_options" not in json.dumps(request_body)
|
|
|
|
|
|
@pytest.mark.usefixtures("_no_openai_api_base_override")
|
|
def test_completion_custom_base_url_sends_no_prompt_cache_breakpoint_for_gpt_5_6():
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key", base_url="http://127.0.0.1:9/v1")
|
|
request_body = _openai_chat_create_kwargs(client, model="gpt-5.6", base_url="http://127.0.0.1:9/v1")
|
|
|
|
assert request_body["messages"][0] == {"role": "system", "content": "sys", "cache_control": {"type": "ephemeral"}}
|
|
assert "prompt_cache_breakpoint" not in json.dumps(request_body["messages"])
|
|
assert "prompt_cache_options" not in json.dumps(request_body)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.usefixtures("_no_openai_api_base_override")
|
|
async def test_acompletion_custom_base_url_sends_no_prompt_cache_breakpoint_for_gpt_5_6():
|
|
from openai import AsyncOpenAI
|
|
|
|
client = AsyncOpenAI(api_key="fake-api-key", base_url="http://127.0.0.1:9/v1")
|
|
with patch.object(client.chat.completions.with_raw_response, "create") as mock_create:
|
|
with contextlib.suppress(Exception):
|
|
await litellm.acompletion(
|
|
model="gpt-5.6",
|
|
messages=[{"role": "system", "content": "sys"}, {"role": "user", "content": "hi"}],
|
|
cache_control_injection_points=[{"location": "message", "role": "system"}],
|
|
client=client,
|
|
base_url="http://127.0.0.1:9/v1",
|
|
)
|
|
|
|
mock_create.assert_called_once()
|
|
request_body = mock_create.call_args.kwargs
|
|
|
|
assert request_body["messages"][0] == {"role": "system", "content": "sys", "cache_control": {"type": "ephemeral"}}
|
|
assert "prompt_cache_breakpoint" not in json.dumps(request_body["messages"])
|
|
assert "prompt_cache_options" not in json.dumps(request_body)
|
|
|
|
|
|
@pytest.mark.usefixtures("_no_openai_api_base_override")
|
|
def test_completion_default_api_base_sends_prompt_cache_breakpoint_for_gpt_5_6():
|
|
from openai import OpenAI
|
|
|
|
client = OpenAI(api_key="fake-api-key")
|
|
request_body = _openai_chat_create_kwargs(client, model="gpt-5.6")
|
|
|
|
assert request_body["messages"][0]["content"] == [
|
|
{"type": "text", "text": "sys", "prompt_cache_breakpoint": {"mode": "explicit"}}
|
|
]
|
|
assert request_body["extra_body"]["prompt_cache_options"] == {"mode": "explicit"}
|
|
|
|
|
|
_SUBSCRIPTION_OAUTH_CREDENTIAL = "Bearer sk-ant-oat01-fake-subscription-token-for-testing-0123456789"
|
|
|
|
|
|
def _scoped_headers_for_oauth_request():
|
|
from litellm.types.utils import ProviderSpecificHeader
|
|
|
|
return [
|
|
ProviderSpecificHeader(
|
|
custom_llm_provider="anthropic,bedrock,vertex_ai",
|
|
extra_headers={"anthropic-version": "2023-06-01"},
|
|
),
|
|
ProviderSpecificHeader(
|
|
custom_llm_provider="anthropic",
|
|
extra_headers={"authorization": _SUBSCRIPTION_OAUTH_CREDENTIAL},
|
|
),
|
|
]
|
|
|
|
|
|
def _run_anthropic_hop_with_shared_headers(shared_headers):
|
|
litellm.completion(
|
|
model="anthropic/claude-3-5-sonnet-20240620",
|
|
messages=[{"role": "user", "content": "Say OK"}],
|
|
extra_headers=shared_headers,
|
|
provider_specific_header=_scoped_headers_for_oauth_request(),
|
|
api_key="sk-fake-anthropic-key",
|
|
mock_response="OK",
|
|
)
|
|
|
|
|
|
def test_completion_does_not_mutate_caller_supplied_headers():
|
|
shared_headers = {"x-tenant": "acme"}
|
|
|
|
_run_anthropic_hop_with_shared_headers(shared_headers)
|
|
|
|
assert shared_headers == {"x-tenant": "acme"}
|
|
|
|
|
|
def test_anthropic_oauth_credential_does_not_persist_into_next_provider_hop():
|
|
shared_headers = {"x-tenant": "acme"}
|
|
|
|
_run_anthropic_hop_with_shared_headers(shared_headers)
|
|
|
|
leaked = [name for name, value in shared_headers.items() if value == _SUBSCRIPTION_OAUTH_CREDENTIAL]
|
|
assert leaked == []
|
|
assert "anthropic-version" not in shared_headers
|
|
|
|
|
|
STREAM_COST_MODEL = "gpt-4o"
|
|
STREAMED_USAGE = {"prompt_tokens": 137, "completion_tokens": 42, "total_tokens": 179}
|
|
|
|
|
|
def _text_chunk(content, finish_reason=None, usage=None):
|
|
chunk = {
|
|
"id": "chatcmpl-stream-cost",
|
|
"object": "chat.completion.chunk",
|
|
"created": 1700000000,
|
|
"model": STREAM_COST_MODEL,
|
|
"choices": [
|
|
{
|
|
"index": 0,
|
|
"delta": {"role": "assistant", "content": content},
|
|
"finish_reason": finish_reason,
|
|
}
|
|
],
|
|
}
|
|
if usage is not None:
|
|
chunk["usage"] = usage
|
|
return chunk
|
|
|
|
|
|
def _priced_at(prompt_tokens, completion_tokens):
|
|
prices = litellm.model_cost[STREAM_COST_MODEL]
|
|
return (
|
|
prompt_tokens * prices["input_cost_per_token"]
|
|
+ completion_tokens * prices["output_cost_per_token"]
|
|
)
|
|
|
|
|
|
@pytest.fixture
|
|
def local_cost_map(monkeypatch):
|
|
"""The prices these tests assert are the checked-in ones. Setting the environment
|
|
variable alone does not reload the map, so pin the map itself.
|
|
|
|
Prices are read through two separate lru_caches, so pinning ``model_cost`` is not
|
|
enough on its own: an entry warmed against the network-fetched map keeps its old
|
|
prices and billing reads those while the assertions read the pinned map.
|
|
``_invalidate_model_cost_lowercase_map`` clears both caches, where
|
|
``get_model_info.cache_clear`` reaches only one. Invalidate on the way in and out
|
|
so entries never leak across tests in either direction."""
|
|
from litellm.utils import _invalidate_model_cost_lowercase_map
|
|
|
|
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
|
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
|
_invalidate_model_cost_lowercase_map()
|
|
yield
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_a_streamed_response_bills_the_usage_the_provider_reported(local_cost_map):
|
|
rebuilt = litellm.stream_chunk_builder(
|
|
chunks=[
|
|
_text_chunk("Hello"),
|
|
_text_chunk(" there"),
|
|
_text_chunk(None, finish_reason="stop", usage=STREAMED_USAGE),
|
|
],
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
)
|
|
|
|
assert rebuilt.choices[0].message.content == "Hello there"
|
|
assert rebuilt.usage.prompt_tokens == STREAMED_USAGE["prompt_tokens"]
|
|
assert rebuilt.usage.completion_tokens == STREAMED_USAGE["completion_tokens"]
|
|
|
|
cost = litellm.completion_cost(completion_response=rebuilt, model=STREAM_COST_MODEL)
|
|
|
|
assert cost == pytest.approx(_priced_at(137, 42))
|
|
assert cost == pytest.approx(0.0007625)
|
|
|
|
|
|
def test_streaming_and_not_streaming_bill_the_same_usage_the_same(local_cost_map):
|
|
rebuilt = litellm.stream_chunk_builder(
|
|
chunks=[
|
|
_text_chunk("Hello"),
|
|
_text_chunk(" there"),
|
|
_text_chunk(None, finish_reason="stop", usage=STREAMED_USAGE),
|
|
],
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
)
|
|
whole = litellm.ModelResponse(
|
|
id="chatcmpl-stream-cost",
|
|
model=STREAM_COST_MODEL,
|
|
object="chat.completion",
|
|
created=1700000000,
|
|
choices=[
|
|
{
|
|
"index": 0,
|
|
"message": {"role": "assistant", "content": "Hello there"},
|
|
"finish_reason": "stop",
|
|
}
|
|
],
|
|
usage=STREAMED_USAGE,
|
|
)
|
|
|
|
assert litellm.completion_cost(
|
|
completion_response=rebuilt, model=STREAM_COST_MODEL
|
|
) == pytest.approx(litellm.completion_cost(completion_response=whole, model=STREAM_COST_MODEL))
|
|
|
|
|
|
def test_a_stream_that_reported_no_usage_is_still_billed(local_cost_map):
|
|
rebuilt = litellm.stream_chunk_builder(
|
|
chunks=[
|
|
_text_chunk("Hello"),
|
|
_text_chunk(" there"),
|
|
_text_chunk(None, finish_reason="stop"),
|
|
],
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
)
|
|
|
|
assert rebuilt.usage.prompt_tokens > 0
|
|
assert rebuilt.usage.completion_tokens > 0
|
|
|
|
cost = litellm.completion_cost(completion_response=rebuilt, model=STREAM_COST_MODEL)
|
|
|
|
assert cost > 0
|
|
assert cost == pytest.approx(
|
|
_priced_at(rebuilt.usage.prompt_tokens, rebuilt.usage.completion_tokens)
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_acompletion_resolves_provider_from_api_base():
|
|
response = await litellm.acompletion(
|
|
model="deepseek-chat",
|
|
api_base="https://api.deepseek.com/v1",
|
|
api_key="fake-key",
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
mock_response="resolved",
|
|
)
|
|
|
|
assert response.choices[0].message.content == "resolved"
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class _RecordedSpeechSuccess:
|
|
call_type: str | None
|
|
spend_metadata: Mapping[str, object]
|
|
response_cost: float | None
|
|
logged_response_cost: float | None
|
|
|
|
|
|
def _record_speech_success(payload: dict[str, object]) -> _RecordedSpeechSuccess:
|
|
call_type: Final = payload.get("call_type")
|
|
response_cost: Final = payload.get("response_cost")
|
|
logging_payload: Final = payload.get("standard_logging_object")
|
|
logged_cost: Final = logging_payload.get("response_cost") if isinstance(logging_payload, dict) else None
|
|
return _RecordedSpeechSuccess(
|
|
call_type=call_type if isinstance(call_type, str) else None,
|
|
spend_metadata=get_litellm_metadata_from_kwargs(payload),
|
|
response_cost=response_cost if isinstance(response_cost, float) else None,
|
|
logged_response_cost=logged_cost if isinstance(logged_cost, float) else None,
|
|
)
|
|
|
|
|
|
class _SuccessEventRecorder(CustomLogger):
|
|
def __init__(self) -> None:
|
|
super().__init__()
|
|
self.events: list[_RecordedSpeechSuccess] = [] # mutable-ok: test recorder of success-callback events
|
|
|
|
async def async_log_success_event(
|
|
self, kwargs: dict[str, object], response_obj: object, start_time: object, end_time: object
|
|
) -> None:
|
|
self.events.append(_record_speech_success(kwargs))
|
|
|
|
|
|
async def _wait_for_success_event(recorder: _SuccessEventRecorder, call_type: str) -> _RecordedSpeechSuccess:
|
|
for _ in range(100):
|
|
if (event := next((e for e in recorder.events if e.call_type == call_type), None)) is not None:
|
|
return event
|
|
await asyncio.sleep(0.05)
|
|
pytest.fail(f"no {call_type} success event; got {[e.call_type for e in recorder.events]}")
|
|
|
|
|
|
def _gemini_tts_generate_content_response() -> dict[str, object]:
|
|
return {
|
|
"candidates": [
|
|
{
|
|
"content": {
|
|
"parts": [
|
|
{
|
|
"inlineData": {
|
|
"mimeType": "audio/L16;codec=pcm;rate=24000",
|
|
"data": base64.b64encode(b"pcm-audio-bytes").decode(),
|
|
}
|
|
}
|
|
],
|
|
"role": "model",
|
|
},
|
|
"finishReason": "STOP",
|
|
"index": 0,
|
|
}
|
|
],
|
|
"usageMetadata": {
|
|
"promptTokenCount": 5,
|
|
"candidatesTokenCount": 60,
|
|
"totalTokenCount": 65,
|
|
"promptTokensDetails": [{"modality": "TEXT", "tokenCount": 5}],
|
|
"candidatesTokensDetails": [{"modality": "AUDIO", "tokenCount": 60}],
|
|
},
|
|
"modelVersion": "gemini-2.5-flash-preview-tts",
|
|
}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_aspeech_gemini_bridge_keeps_proxy_metadata_for_spend_tracking(
|
|
respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
|
monkeypatch.delenv("GEMINI_API_KEY", raising=False)
|
|
monkeypatch.delenv("GOOGLE_API_KEY", raising=False)
|
|
recorder: Final = _SuccessEventRecorder()
|
|
monkeypatch.setattr(litellm, "callbacks", [recorder])
|
|
mock_route: Final = respx_mock.post(
|
|
url__regex=r"https://generativelanguage\.googleapis\.com/v1beta/models/gemini-2\.5-flash-preview-tts:generateContent.*"
|
|
).mock(return_value=httpx.Response(200, json=_gemini_tts_generate_content_response()))
|
|
|
|
await litellm.aspeech(
|
|
model="gemini/gemini-2.5-flash-preview-tts",
|
|
input="spend tracking check",
|
|
voice="Kore",
|
|
api_key="fake-gemini-key",
|
|
metadata={"user_api_key": "hashed-virtual-key", "user_api_key_user_id": "user-1"},
|
|
)
|
|
|
|
assert mock_route.called
|
|
assert mock_route.calls.last.request.headers["x-goog-api-key"] == "fake-gemini-key"
|
|
speech_event: Final = await _wait_for_success_event(recorder, call_type="aspeech")
|
|
assert speech_event.spend_metadata["user_api_key"] == "hashed-virtual-key"
|
|
assert speech_event.spend_metadata["user_api_key_user_id"] == "user-1"
|
|
expected_prompt_cost, expected_completion_cost = litellm.cost_per_token(
|
|
model="gemini/gemini-2.5-flash-preview-tts",
|
|
usage_object=Usage(prompt_tokens=5, completion_tokens=60, total_tokens=65),
|
|
)
|
|
expected_cost: Final = expected_prompt_cost + expected_completion_cost
|
|
assert expected_cost > 0
|
|
assert speech_event.response_cost == pytest.approx(expected_cost)
|
|
assert speech_event.logged_response_cost == pytest.approx(expected_cost)
|
|
|
|
|
|
def _stream_builder_text_chunk(model: str, content: str, finish_reason: str | None = None) -> ModelResponseStream:
|
|
return ModelResponseStream(
|
|
id="chatcmpl-cost",
|
|
created=1724900000,
|
|
model=model,
|
|
object="chat.completion.chunk",
|
|
choices=[StreamingChoices(finish_reason=finish_reason, index=0, delta=Delta(content=content, role="assistant"))],
|
|
)
|
|
|
|
|
|
def test_stream_chunk_builder_sets_hidden_response_cost_for_known_model():
|
|
chunks: Final = [
|
|
_stream_builder_text_chunk("gpt-4o", "Hello "),
|
|
_stream_builder_text_chunk("gpt-4o", "world.", finish_reason="stop"),
|
|
]
|
|
|
|
response: Final = litellm.stream_chunk_builder(chunks=chunks, messages=[{"role": "user", "content": "hi"}])
|
|
|
|
assert response is not None
|
|
prompt_cost, completion_cost = litellm.cost_per_token(model="gpt-4o", usage_object=response.usage)
|
|
expected_cost: Final = prompt_cost + completion_cost
|
|
assert expected_cost > 0
|
|
assert response._hidden_params["response_cost"] == pytest.approx(expected_cost)
|
|
|
|
|
|
def test_stream_chunk_builder_unknown_model_leaves_response_cost_unset():
|
|
chunks: Final = [
|
|
_stream_builder_text_chunk("totally-unknown-model-xyz", "Hello "),
|
|
_stream_builder_text_chunk("totally-unknown-model-xyz", "world.", finish_reason="stop"),
|
|
]
|
|
|
|
response: Final = litellm.stream_chunk_builder(chunks=chunks, messages=[{"role": "user", "content": "hi"}])
|
|
|
|
assert response is not None
|
|
assert response._hidden_params.get("response_cost") is None
|
|
assert response.choices[0].message.content == "Hello world."
|
|
|
|
|
|
def test_stream_chunk_builder_prices_proxy_alias_via_model_map():
|
|
chunks: Final = [
|
|
_stream_builder_text_chunk("claude-opus-5", "Hello "),
|
|
_stream_builder_text_chunk("claude-opus-5", "world.", finish_reason="stop"),
|
|
]
|
|
for chunk in chunks:
|
|
chunk._hidden_params = {"custom_llm_provider": "openai"}
|
|
|
|
response: Final = litellm.stream_chunk_builder(chunks=chunks, messages=[{"role": "user", "content": "hi"}])
|
|
|
|
assert response is not None
|
|
assert response._hidden_params["custom_llm_provider"] == "openai"
|
|
prompt_cost, completion_cost = litellm.cost_per_token(model="claude-opus-5", usage_object=response.usage)
|
|
expected_cost: Final = prompt_cost + completion_cost
|
|
assert expected_cost > 0
|
|
assert response._hidden_params["response_cost"] == pytest.approx(expected_cost)
|
|
|
|
|
|
def _stream_builder_logging_obj(model: str = "gpt-4o", custom_llm_provider: str = "openai") -> LiteLLMLogging:
|
|
logging_obj: Final = LiteLLMLogging(
|
|
model=model,
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
stream=True,
|
|
call_type="completion",
|
|
start_time=datetime.now(),
|
|
litellm_call_id="test-call-id",
|
|
function_id="test-function-id",
|
|
)
|
|
logging_obj.update_environment_variables(
|
|
model=model,
|
|
user=None,
|
|
optional_params={},
|
|
litellm_params={"custom_llm_provider": custom_llm_provider},
|
|
custom_llm_provider=custom_llm_provider,
|
|
)
|
|
return logging_obj
|
|
|
|
|
|
def test_stream_chunk_builder_stamps_streaming_usage_cost_by_default(monkeypatch: pytest.MonkeyPatch):
|
|
monkeypatch.setattr(litellm, "include_cost_in_streaming_usage", False)
|
|
chunks: Final = [
|
|
_stream_builder_text_chunk("gpt-4o", "Hello "),
|
|
_stream_builder_text_chunk("gpt-4o", "world.", finish_reason="stop"),
|
|
]
|
|
|
|
response: Final = litellm.stream_chunk_builder(
|
|
chunks=chunks, messages=[{"role": "user", "content": "hi"}], logging_obj=_stream_builder_logging_obj()
|
|
)
|
|
|
|
assert response is not None
|
|
usage_cost: Final = getattr(response.usage, "cost", None)
|
|
assert usage_cost is not None
|
|
assert usage_cost > 0
|
|
assert response._hidden_params["response_cost"] == pytest.approx(usage_cost)
|
|
|
|
|
|
def test_stream_chunk_builder_skips_stamp_when_cost_is_unpriceable():
|
|
import time as time_module
|
|
|
|
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLogging
|
|
|
|
logging_obj: Final = LiteLLMLogging(
|
|
model="us.anthropic.claude-opus-5",
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
stream=True,
|
|
call_type="completion",
|
|
start_time=time_module.time(),
|
|
litellm_call_id="stream-builder-alias-unpriceable",
|
|
function_id="1",
|
|
)
|
|
logging_obj.model_call_details["custom_llm_provider"] = "bedrock"
|
|
logging_obj.optional_params = {}
|
|
usage_chunk: Final = _stream_builder_text_chunk("bedrock-claude-opus-5", "")
|
|
usage_chunk.usage = Usage(prompt_tokens=40, completion_tokens=5, total_tokens=45)
|
|
chunks: Final = [
|
|
_stream_builder_text_chunk("bedrock-claude-opus-5", "Hello ", finish_reason="stop"),
|
|
usage_chunk,
|
|
]
|
|
|
|
response: Final = litellm.stream_chunk_builder(
|
|
chunks=chunks, messages=[{"role": "user", "content": "hi"}], logging_obj=logging_obj
|
|
)
|
|
|
|
assert response is not None
|
|
assert getattr(response.usage, "cost", None) is None
|
|
assert response._hidden_params.get("response_cost") is None
|
|
|
|
|
|
def test_stream_chunk_builder_keeps_provider_reported_usage_cost():
|
|
usage_chunk: Final = _stream_builder_text_chunk("gpt-4o", "")
|
|
usage_chunk.usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15, cost=0.5)
|
|
chunks: Final = [
|
|
_stream_builder_text_chunk("gpt-4o", "Hello "),
|
|
_stream_builder_text_chunk("gpt-4o", "world.", finish_reason="stop"),
|
|
usage_chunk,
|
|
]
|
|
|
|
response: Final = litellm.stream_chunk_builder(
|
|
chunks=chunks, messages=[{"role": "user", "content": "hi"}], logging_obj=_stream_builder_logging_obj()
|
|
)
|
|
|
|
assert response is not None
|
|
assert getattr(response.usage, "cost", None) == pytest.approx(0.5)
|
|
assert response._hidden_params["response_cost"] == pytest.approx(0.5)
|
|
|
|
|
|
def test_stream_chunk_builder_prices_alias_from_openai_sdk_usage_chunk():
|
|
from openai.types.completion_usage import CompletionUsage
|
|
|
|
usage_chunk: Final = _stream_builder_text_chunk("mantle-claude", "")
|
|
usage_chunk.usage = CompletionUsage(prompt_tokens=20, completion_tokens=60, total_tokens=80, cost=0.000704)
|
|
assert type(usage_chunk.usage) is CompletionUsage
|
|
chunks: Final = [
|
|
_stream_builder_text_chunk("mantle-claude", "Hello "),
|
|
_stream_builder_text_chunk("mantle-claude", "world.", finish_reason="stop"),
|
|
usage_chunk,
|
|
]
|
|
|
|
response: Final = litellm.stream_chunk_builder(chunks=chunks, messages=[{"role": "user", "content": "hi"}])
|
|
|
|
assert response is not None
|
|
assert response.usage.prompt_tokens == 20
|
|
assert response.usage.completion_tokens == 60
|
|
assert getattr(response.usage, "cost", None) == pytest.approx(0.000704)
|
|
assert response._hidden_params["response_cost"] == pytest.approx(0.000704)
|
|
|
|
|
|
def test_stream_chunk_builder_leaves_xai_reported_cost_to_the_calculator(monkeypatch: pytest.MonkeyPatch):
|
|
monkeypatch.setattr(litellm, "cost_margin_config", {"xai": 0.5})
|
|
usage_chunk: Final = _stream_builder_text_chunk("grok-4", "")
|
|
usage_chunk.usage = Usage(prompt_tokens=5, completion_tokens=2, total_tokens=7, cost=0.42)
|
|
chunks: Final = [
|
|
_stream_builder_text_chunk("grok-4", "Hello "),
|
|
_stream_builder_text_chunk("grok-4", "world.", finish_reason="stop"),
|
|
usage_chunk,
|
|
]
|
|
logging_obj: Final = _stream_builder_logging_obj(model="grok-4", custom_llm_provider="xai")
|
|
|
|
response: Final = litellm.stream_chunk_builder(
|
|
chunks=chunks, messages=[{"role": "user", "content": "hi"}], logging_obj=logging_obj
|
|
)
|
|
|
|
assert response is not None
|
|
assert getattr(response.usage, "cost", None) == pytest.approx(0.42)
|
|
assert response._hidden_params.get("response_cost") is None
|
|
assert logging_obj._response_cost_calculator(result=response) == pytest.approx(0.63)
|
|
|
|
|
|
def test_speech_mistral_dispatches_and_decodes_audio(respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch):
|
|
monkeypatch.setenv("MISTRAL_API_KEY", "sk-mistral-test")
|
|
audio_bytes: Final = b"ID3-fake-mp3-bytes"
|
|
mock_route: Final = respx_mock.post("https://api.mistral.ai/v1/audio/speech").mock(
|
|
return_value=httpx.Response(200, json={"audio_data": base64.b64encode(audio_bytes).decode()})
|
|
)
|
|
|
|
response: Final = litellm.speech(
|
|
model="mistral/voxtral-mini-tts-2603",
|
|
input="hello from litellm",
|
|
voice="en_paul_neutral",
|
|
response_format="wav",
|
|
speed=2,
|
|
instructions="sound cheerful",
|
|
)
|
|
|
|
assert mock_route.called
|
|
request_body: Final = json.loads(mock_route.calls.last.request.content)
|
|
assert request_body == {
|
|
"model": "voxtral-mini-tts-2603",
|
|
"input": "hello from litellm",
|
|
"voice_id": "en_paul_neutral",
|
|
"response_format": "wav",
|
|
}
|
|
assert mock_route.calls.last.request.headers["authorization"] == "Bearer sk-mistral-test"
|
|
assert response.content == audio_bytes
|
|
|
|
|
|
def test_speech_mistral_routes_to_configured_api_base(respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch):
|
|
monkeypatch.setenv("MISTRAL_API_KEY", "sk-mistral-test")
|
|
audio_bytes: Final = b"ID3-gateway-bytes"
|
|
gateway_route: Final = respx_mock.post("https://mistral.gateway.internal/v1/audio/speech").mock(
|
|
return_value=httpx.Response(200, json={"audio_data": base64.b64encode(audio_bytes).decode()})
|
|
)
|
|
|
|
response: Final = litellm.speech(
|
|
model="mistral/voxtral-mini-tts-2603",
|
|
input="hello from litellm",
|
|
voice="en_paul_neutral",
|
|
api_base="https://mistral.gateway.internal",
|
|
)
|
|
|
|
assert gateway_route.called
|
|
assert response.content == audio_bytes
|
|
|
|
|
|
FOUNDRY_HOST: Final = "https://my-project.services.ai.azure.com"
|
|
|
|
|
|
def test_azure_ai_transcription_on_a_foundry_host_uses_the_azure_openai_deployment_route(
|
|
respx_mock: respx.MockRouter,
|
|
):
|
|
route: Final = respx_mock.post(
|
|
url__regex=r"https://my-project\.services\.ai\.azure\.com/openai/deployments/whisper-1/audio/transcriptions\?api-version=.+"
|
|
).mock(return_value=httpx.Response(200, json={"text": "hello"}))
|
|
|
|
response: Final = litellm.transcription(
|
|
model="azure_ai/whisper-1",
|
|
file=("tone.wav", b"RIFF\x00\x00\x00\x00WAVE", "audio/wav"),
|
|
api_base=FOUNDRY_HOST,
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert route.called
|
|
assert response.text == "hello"
|
|
|
|
|
|
def test_azure_ai_speech_on_a_foundry_host_uses_the_azure_openai_deployment_route(
|
|
respx_mock: respx.MockRouter,
|
|
):
|
|
route: Final = respx_mock.post(
|
|
url__regex=r"https://my-project\.services\.ai\.azure\.com/openai/deployments/tts-1/audio/speech\?api-version=.+"
|
|
).mock(return_value=httpx.Response(200, content=b"mp3-bytes"))
|
|
|
|
response: Final = litellm.speech(
|
|
model="azure_ai/tts-1",
|
|
input="hello",
|
|
voice="alloy",
|
|
api_base=FOUNDRY_HOST,
|
|
api_key="fake-key",
|
|
)
|
|
|
|
assert route.called
|
|
assert response.content == b"mp3-bytes"
|