mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
test(ci): refresh qualified retired OpenAI fixtures (#43938)
* test(ci): refresh qualified retired OpenAI fixtures * test(ci): compare fallback input usage instead of provider wording
This commit is contained in:
parent
c42d06fb80
commit
f8f05767da
5 changed files with 167 additions and 201 deletions
|
|
@ -141,6 +141,11 @@ model_list:
|
|||
- model_name: mistral-embed
|
||||
litellm_params:
|
||||
model: mistral/mistral-embed
|
||||
- model_name: gpt-6-luna
|
||||
litellm_params:
|
||||
model: openai/gpt-6-luna
|
||||
reasoning_effort: none
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: gpt-instruct # [PROD TEST] - tests if `/health` automatically infers this to be a text completion model
|
||||
litellm_params:
|
||||
model: text-completion-openai/gpt-3.5-turbo-instruct
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@
|
|||
# This tests streaming for the completion endpoint
|
||||
|
||||
import asyncio
|
||||
from typing import Final
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
|
|
@ -1546,45 +1547,24 @@ async def test_openai_stream_options_call(model, sync):
|
|||
)
|
||||
|
||||
|
||||
def test_openai_stream_options_call_text_completion():
|
||||
litellm.set_verbose = False
|
||||
for idx in range(3):
|
||||
try:
|
||||
response = litellm.text_completion(
|
||||
model="gpt-3.5-turbo-instruct",
|
||||
prompt="say GM - we're going to make it ",
|
||||
stream=True,
|
||||
stream_options={"include_usage": True},
|
||||
max_tokens=10,
|
||||
)
|
||||
usage = None
|
||||
chunks = []
|
||||
for chunk in response:
|
||||
print("chunk: ", chunk)
|
||||
chunks.append(chunk)
|
||||
|
||||
last_chunk = chunks[-1]
|
||||
print("last chunk: ", last_chunk)
|
||||
|
||||
"""
|
||||
Assert that:
|
||||
- Last Chunk includes Usage
|
||||
- All chunks prior to last chunk have usage=None
|
||||
"""
|
||||
|
||||
assert last_chunk.usage is not None
|
||||
assert last_chunk.usage.total_tokens > 0
|
||||
assert last_chunk.usage.prompt_tokens > 0
|
||||
assert last_chunk.usage.completion_tokens > 0
|
||||
|
||||
# assert all non last chunks have usage=None
|
||||
assert all(chunk.usage is None for chunk in chunks[:-1])
|
||||
break
|
||||
except Exception as e:
|
||||
if idx < 2:
|
||||
pass
|
||||
else:
|
||||
raise e
|
||||
def test_openai_stream_options_call_text_completion() -> None:
|
||||
chunks: Final = tuple(
|
||||
litellm.text_completion(
|
||||
model="gpt-6-luna",
|
||||
reasoning_effort="none",
|
||||
prompt="say GM - we're going to make it ",
|
||||
stream=True,
|
||||
stream_options={"include_usage": True},
|
||||
max_tokens=10,
|
||||
)
|
||||
)
|
||||
assert chunks
|
||||
assert chunks[-1].usage is not None
|
||||
assert chunks[-1].usage.total_tokens > 0
|
||||
assert chunks[-1].usage.prompt_tokens > 0
|
||||
assert chunks[-1].usage.completion_tokens > 0
|
||||
assert all(chunk.usage is None for chunk in chunks[:-1])
|
||||
assert any(chunk.choices[0].text for chunk in chunks)
|
||||
|
||||
|
||||
def test_openai_text_completion_call():
|
||||
|
|
@ -1676,8 +1656,8 @@ def test_together_ai_completion_call_starcoder_bad_key():
|
|||
#### Test Function calling + streaming ####
|
||||
|
||||
|
||||
def test_completion_openai_with_functions():
|
||||
function1 = [
|
||||
def test_completion_openai_with_functions() -> None:
|
||||
functions: Final = [
|
||||
{
|
||||
"name": "get_current_weather",
|
||||
"description": "Get the current weather in a given location",
|
||||
|
|
@ -1694,24 +1674,25 @@ def test_completion_openai_with_functions():
|
|||
},
|
||||
}
|
||||
]
|
||||
try:
|
||||
litellm.set_verbose = False
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo-1106",
|
||||
messages=[{"role": "user", "content": "what's the weather in SF"}],
|
||||
functions=function1,
|
||||
messages: Final = [{"role": "user", "content": "what's the weather in SF"}]
|
||||
chunks: Final = tuple(
|
||||
completion(
|
||||
model="gpt-6-luna",
|
||||
reasoning_effort="none",
|
||||
messages=messages,
|
||||
functions=functions,
|
||||
function_call={"name": "get_current_weather"},
|
||||
stream=True,
|
||||
max_tokens=128,
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
print(response)
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
if chunk["choices"][0]["finish_reason"] == "stop":
|
||||
break
|
||||
print(chunk["choices"][0]["finish_reason"])
|
||||
print(chunk["choices"][0]["delta"]["content"])
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
)
|
||||
response: Final = litellm.stream_chunk_builder(chunks, messages=messages)
|
||||
assert response is not None
|
||||
function_call: Final = response.choices[0].message.function_call
|
||||
assert function_call is not None
|
||||
assert function_call.name == "get_current_weather"
|
||||
assert json.loads(function_call.arguments)["location"]
|
||||
assert sum(chunk.choices[0].finish_reason is not None for chunk in chunks) == 1
|
||||
|
||||
|
||||
#### Test Async streaming ####
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import asyncio
|
||||
from typing import Final
|
||||
import json
|
||||
import traceback
|
||||
|
||||
|
|
@ -3790,42 +3791,30 @@ def test_completion_openai_prompt():
|
|||
# test_completion_openai_prompt()
|
||||
|
||||
|
||||
def test_completion_openai_engine_and_model():
|
||||
try:
|
||||
print("\n text 003 test\n")
|
||||
litellm.set_verbose = True
|
||||
response = text_completion(
|
||||
model="gpt-3.5-turbo-instruct",
|
||||
engine="anything",
|
||||
prompt="What's the weather in SF?",
|
||||
max_tokens=5,
|
||||
)
|
||||
print(response)
|
||||
response_str = response["choices"][0]["text"]
|
||||
# print(response.choices[0])
|
||||
# print(response.choices[0].text)
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
def test_completion_openai_engine_and_model() -> None:
|
||||
response: Final = text_completion(
|
||||
model="gpt-6-luna",
|
||||
engine="anything",
|
||||
reasoning_effort="none",
|
||||
prompt="What's the weather in SF?",
|
||||
max_tokens=5,
|
||||
)
|
||||
assert response.model == "gpt-6-luna"
|
||||
assert response.choices[0].text
|
||||
|
||||
|
||||
# test_completion_openai_engine_and_model()
|
||||
|
||||
|
||||
def test_completion_openai_engine():
|
||||
try:
|
||||
print("\n text 003 test\n")
|
||||
litellm.set_verbose = True
|
||||
response = text_completion(
|
||||
engine="gpt-3.5-turbo-instruct",
|
||||
prompt="What's the weather in SF?",
|
||||
max_tokens=5,
|
||||
)
|
||||
print(response)
|
||||
response_str = response["choices"][0]["text"]
|
||||
# print(response.choices[0])
|
||||
# print(response.choices[0].text)
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
def test_completion_openai_engine() -> None:
|
||||
response: Final = text_completion(
|
||||
engine="gpt-6-luna",
|
||||
reasoning_effort="none",
|
||||
prompt="What's the weather in SF?",
|
||||
max_tokens=5,
|
||||
)
|
||||
assert response.model == "gpt-6-luna"
|
||||
assert response.choices[0].text
|
||||
|
||||
|
||||
# test_completion_openai_engine()
|
||||
|
|
@ -4048,34 +4037,18 @@ def test_async_text_completion_together_ai():
|
|||
# test_async_text_completion()
|
||||
|
||||
|
||||
def test_async_text_completion_stream():
|
||||
# tests atext_completion + streaming - assert only one finish reason sent
|
||||
litellm.set_verbose = False
|
||||
print("test_async_text_completion with stream")
|
||||
|
||||
async def test_get_response():
|
||||
try:
|
||||
response = await litellm.atext_completion(
|
||||
model="gpt-3.5-turbo-instruct",
|
||||
prompt="good morning",
|
||||
stream=True,
|
||||
)
|
||||
print(f"response: {response}")
|
||||
|
||||
num_finish_reason = 0
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
if chunk["choices"][0].get("finish_reason") is not None:
|
||||
num_finish_reason += 1
|
||||
print("finish_reason", chunk["choices"][0].get("finish_reason"))
|
||||
|
||||
assert (
|
||||
num_finish_reason == 1
|
||||
), f"expected only one finish reason. Got {num_finish_reason}"
|
||||
except Exception as e:
|
||||
pytest.fail(f"GOT exception for gpt-3.5 instruct In streaming{e}")
|
||||
|
||||
asyncio.run(test_get_response())
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_text_completion_stream() -> None:
|
||||
response: Final = await litellm.atext_completion(
|
||||
model="gpt-6-luna",
|
||||
reasoning_effort="none",
|
||||
prompt="good morning",
|
||||
stream=True,
|
||||
max_tokens=32,
|
||||
)
|
||||
chunks: Final = [chunk async for chunk in response]
|
||||
assert sum(chunk.choices[0].finish_reason is not None for chunk in chunks) == 1
|
||||
assert any(chunk.choices[0].text for chunk in chunks)
|
||||
|
||||
|
||||
# test_async_text_completion_stream()
|
||||
|
|
|
|||
|
|
@ -1,3 +1,6 @@
|
|||
import os
|
||||
from typing import Final
|
||||
|
||||
# What is this?
|
||||
## This tests if the proxy fallbacks work as expected
|
||||
import pytest
|
||||
|
|
@ -6,6 +9,9 @@ import aiohttp
|
|||
from tests.large_text import text
|
||||
import time
|
||||
from typing import Optional
|
||||
from openai import AsyncOpenAI, PermissionDeniedError
|
||||
|
||||
PROXY_BASE_URL: Final = os.environ.get("LITELLM_PROXY_BASE_URL", "http://0.0.0.0:4000")
|
||||
|
||||
|
||||
async def generate_key(
|
||||
|
|
@ -14,7 +20,7 @@ async def generate_key(
|
|||
models: list,
|
||||
calling_key="sk-1234",
|
||||
):
|
||||
url = "http://0.0.0.0:4000/key/generate"
|
||||
url: Final = f"{PROXY_BASE_URL}/key/generate"
|
||||
headers = {
|
||||
"Authorization": f"Bearer {calling_key}",
|
||||
"Content-Type": "application/json",
|
||||
|
|
@ -48,7 +54,7 @@ async def chat_completion(
|
|||
extra_headers: Optional[dict] = None,
|
||||
**kwargs,
|
||||
):
|
||||
url = "http://0.0.0.0:4000/chat/completions"
|
||||
url: Final = f"{PROXY_BASE_URL}/chat/completions"
|
||||
headers = {
|
||||
"Authorization": f"Bearer {key}",
|
||||
"Content-Type": "application/json",
|
||||
|
|
@ -94,42 +100,30 @@ async def test_chat_completion():
|
|||
|
||||
@pytest.mark.parametrize("has_access", [True, False])
|
||||
@pytest.mark.asyncio
|
||||
async def test_chat_completion_client_fallbacks(has_access):
|
||||
"""
|
||||
make chat completion call with prompt > context window. expect it to work with fallback
|
||||
"""
|
||||
|
||||
async def test_chat_completion_client_fallbacks(has_access: bool) -> None:
|
||||
models: Final = ["gpt-3.5-turbo", "gpt-6-luna"] if has_access else ["gpt-3.5-turbo"]
|
||||
async with aiohttp.ClientSession() as session:
|
||||
models = ["gpt-3.5-turbo"]
|
||||
|
||||
if has_access:
|
||||
models.append("gpt-instruct")
|
||||
|
||||
## CREATE KEY WITH MODELS
|
||||
generated_key = await generate_key(session=session, i=0, models=models)
|
||||
calling_key = generated_key["key"]
|
||||
model = "gpt-3.5-turbo"
|
||||
messages = [
|
||||
{"role": "user", "content": "Who was Alexander?"},
|
||||
]
|
||||
|
||||
## CALL PROXY
|
||||
try:
|
||||
await chat_completion(
|
||||
session=session,
|
||||
key=calling_key,
|
||||
model=model,
|
||||
messages=messages,
|
||||
mock_testing_fallbacks=True,
|
||||
fallbacks=["gpt-instruct"],
|
||||
)
|
||||
if not has_access:
|
||||
pytest.fail(
|
||||
"Expected this to fail, submitted fallback model that key did not have access to"
|
||||
)
|
||||
except Exception as e:
|
||||
if has_access:
|
||||
pytest.fail("Expected this to work: {}".format(str(e)))
|
||||
generated_key: Final = await generate_key(session=session, i=0, models=models)
|
||||
async with AsyncOpenAI(api_key=generated_key["key"], base_url=PROXY_BASE_URL, max_retries=0) as client:
|
||||
request: Final = {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [{"role": "user", "content": "Who was Alexander?"}],
|
||||
"max_tokens": 32,
|
||||
"temperature": 0,
|
||||
"extra_body": {
|
||||
"mock_testing_fallbacks": True,
|
||||
"fallbacks": ["gpt-6-luna"],
|
||||
},
|
||||
}
|
||||
if not has_access:
|
||||
with pytest.raises(PermissionDeniedError) as denied:
|
||||
await client.chat.completions.create(**request)
|
||||
assert denied.value.status_code == 403
|
||||
assert "gpt-6-luna" in str(denied.value)
|
||||
return
|
||||
response: Final = await client.chat.completions.create(**request)
|
||||
assert response.model == "gpt-6-luna"
|
||||
assert response.choices[0].message.content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -241,55 +235,66 @@ async def test_chat_completion_with_timeout_from_request():
|
|||
|
||||
@pytest.mark.parametrize("has_access", [True, False])
|
||||
@pytest.mark.asyncio
|
||||
async def test_chat_completion_client_fallbacks_with_custom_message(has_access):
|
||||
"""
|
||||
make chat completion call with prompt > context window. expect it to work with fallback
|
||||
"""
|
||||
|
||||
async def test_chat_completion_client_fallbacks_with_custom_message(has_access: bool) -> None:
|
||||
original_messages: Final = [{"role": "user", "content": "Who was Alexander?"}]
|
||||
custom_messages: Final = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": (
|
||||
"Describe the weather in a coastal city during winter, including the usual temperature, rain, wind, "
|
||||
"and the clothing a visitor should bring."
|
||||
),
|
||||
}
|
||||
]
|
||||
models: Final = ["gpt-3.5-turbo", "gpt-6-luna"] if has_access else ["gpt-3.5-turbo"]
|
||||
async with aiohttp.ClientSession() as session:
|
||||
models = ["gpt-3.5-turbo"]
|
||||
|
||||
if has_access:
|
||||
models.append("gpt-instruct")
|
||||
|
||||
## CREATE KEY WITH MODELS
|
||||
generated_key = await generate_key(session=session, i=0, models=models)
|
||||
calling_key = generated_key["key"]
|
||||
model = "gpt-3.5-turbo"
|
||||
messages = [
|
||||
{"role": "user", "content": "Who was Alexander?"},
|
||||
]
|
||||
|
||||
## CALL PROXY
|
||||
try:
|
||||
await chat_completion(
|
||||
session=session,
|
||||
key=calling_key,
|
||||
model=model,
|
||||
messages=messages,
|
||||
mock_testing_fallbacks=True,
|
||||
fallbacks=[
|
||||
generated_key: Final = await generate_key(session=session, i=0, models=models)
|
||||
async with AsyncOpenAI(api_key=generated_key["key"], base_url=PROXY_BASE_URL, max_retries=0) as client:
|
||||
request: Final = {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": original_messages,
|
||||
"max_tokens": 32,
|
||||
"temperature": 0,
|
||||
"extra_body": {
|
||||
"mock_testing_fallbacks": True,
|
||||
"fallbacks": [
|
||||
{
|
||||
"model": "gpt-instruct",
|
||||
"messages": [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "This is a custom message",
|
||||
}
|
||||
],
|
||||
"model": "gpt-6-luna",
|
||||
"messages": custom_messages,
|
||||
}
|
||||
],
|
||||
)
|
||||
if not has_access:
|
||||
pytest.fail(
|
||||
"Expected this to fail, submitted fallback model that key did not have access to"
|
||||
)
|
||||
except Exception as e:
|
||||
if has_access:
|
||||
pytest.fail("Expected this to work: {}".format(str(e)))
|
||||
},
|
||||
}
|
||||
if not has_access:
|
||||
with pytest.raises(PermissionDeniedError) as denied:
|
||||
await client.chat.completions.create(**request)
|
||||
assert denied.value.status_code == 403
|
||||
assert "gpt-6-luna" in str(denied.value)
|
||||
return
|
||||
response: Final = await client.chat.completions.create(**request)
|
||||
assert response.model == "gpt-6-luna"
|
||||
assert response.choices[0].message.content
|
||||
custom_control: Final = await client.chat.completions.create(
|
||||
model="gpt-6-luna",
|
||||
messages=custom_messages,
|
||||
max_tokens=32,
|
||||
temperature=0,
|
||||
)
|
||||
original_control: Final = await client.chat.completions.create(
|
||||
model="gpt-6-luna",
|
||||
messages=original_messages,
|
||||
max_tokens=32,
|
||||
temperature=0,
|
||||
)
|
||||
assert response.usage is not None
|
||||
assert custom_control.usage is not None
|
||||
assert original_control.usage is not None
|
||||
assert custom_control.usage.completion_tokens > 0
|
||||
assert original_control.usage.completion_tokens > 0
|
||||
assert custom_control.usage.prompt_tokens != original_control.usage.prompt_tokens
|
||||
assert response.usage.prompt_tokens == custom_control.usage.prompt_tokens
|
||||
|
||||
|
||||
from openai import AsyncOpenAI
|
||||
from typing import List
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
import os
|
||||
from typing import Final
|
||||
# What this tests ?
|
||||
## Tests /chat/completions by generating a key and then making a chat completions-request
|
||||
import pytest
|
||||
|
|
@ -398,10 +400,12 @@ async def test_completion_streaming_usage_metrics():
|
|||
"""
|
||||
[PROD Test] Ensures usage metrics are returned correctly when `include_usage` is set to `True`
|
||||
"""
|
||||
client = AsyncOpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000")
|
||||
client: Final = AsyncOpenAI(
|
||||
api_key="sk-1234", base_url=os.environ.get("LITELLM_PROXY_BASE_URL", "http://0.0.0.0:4000")
|
||||
)
|
||||
|
||||
response = await client.completions.create(
|
||||
model="gpt-instruct",
|
||||
model="gpt-6-luna",
|
||||
prompt="hey",
|
||||
stream=True,
|
||||
stream_options={"include_usage": True},
|
||||
|
|
@ -417,9 +421,7 @@ async def test_completion_streaming_usage_metrics():
|
|||
assert last_chunk is not None, "No chunks were received"
|
||||
assert last_chunk.usage is not None, "Usage information was not received"
|
||||
assert last_chunk.usage.prompt_tokens > 0, "Prompt tokens should be greater than 0"
|
||||
assert (
|
||||
last_chunk.usage.completion_tokens > 0
|
||||
), "Completion tokens should be greater than 0"
|
||||
assert last_chunk.usage.completion_tokens > 0, "Completion tokens should be greater than 0"
|
||||
assert last_chunk.usage.total_tokens > 0, "Total tokens should be greater than 0"
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue