test(ci): refresh qualified retired OpenAI fixtures (#43938)

* test(ci): refresh qualified retired OpenAI fixtures

* test(ci): compare fallback input usage instead of provider wording
This commit is contained in:
yuneng-jiang 2026-09-30 16:10:18 -07:00 • committed by GitHub
parent c42d06fb80
commit f8f05767da
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 167 additions and 201 deletions

View file

@ -141,6 +141,11 @@ model_list:
- model_name: mistral-embed
litellm_params:
model: mistral/mistral-embed
- model_name: gpt-6-luna
litellm_params:
model: openai/gpt-6-luna
reasoning_effort: none
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-instruct # [PROD TEST] - tests if `/health` automatically infers this to be a text completion model
litellm_params:
model: text-completion-openai/gpt-3.5-turbo-instruct

View file

@ -2,6 +2,7 @@
# This tests streaming for the completion endpoint
import asyncio
from typing import Final
import json
import os
import time
@ -1546,45 +1547,24 @@ async def test_openai_stream_options_call(model, sync):
)
def test_openai_stream_options_call_text_completion():
litellm.set_verbose = False
for idx in range(3):
try:
response = litellm.text_completion(
model="gpt-3.5-turbo-instruct",
prompt="say GM - we're going to make it ",
stream=True,
stream_options={"include_usage": True},
max_tokens=10,
)
usage = None
chunks = []
for chunk in response:
print("chunk: ", chunk)
chunks.append(chunk)
last_chunk = chunks[-1]
print("last chunk: ", last_chunk)
"""
Assert that:
- Last Chunk includes Usage
- All chunks prior to last chunk have usage=None
"""
assert last_chunk.usage is not None
assert last_chunk.usage.total_tokens > 0
assert last_chunk.usage.prompt_tokens > 0
assert last_chunk.usage.completion_tokens > 0
# assert all non last chunks have usage=None
assert all(chunk.usage is None for chunk in chunks[:-1])
break
except Exception as e:
if idx < 2:
pass
else:
raise e
def test_openai_stream_options_call_text_completion() -> None:
chunks: Final = tuple(
litellm.text_completion(
model="gpt-6-luna",
reasoning_effort="none",
prompt="say GM - we're going to make it ",
stream=True,
stream_options={"include_usage": True},
max_tokens=10,
)
)
assert chunks
assert chunks[-1].usage is not None
assert chunks[-1].usage.total_tokens > 0
assert chunks[-1].usage.prompt_tokens > 0
assert chunks[-1].usage.completion_tokens > 0
assert all(chunk.usage is None for chunk in chunks[:-1])
assert any(chunk.choices[0].text for chunk in chunks)
def test_openai_text_completion_call():
@ -1676,8 +1656,8 @@ def test_together_ai_completion_call_starcoder_bad_key():
#### Test Function calling + streaming ####
def test_completion_openai_with_functions():
function1 = [
def test_completion_openai_with_functions() -> None:
functions: Final = [
{
"name": "get_current_weather",
"description": "Get the current weather in a given location",
@ -1694,24 +1674,25 @@ def test_completion_openai_with_functions():
},
}
]
try:
litellm.set_verbose = False
response = completion(
model="gpt-3.5-turbo-1106",
messages=[{"role": "user", "content": "what's the weather in SF"}],
functions=function1,
messages: Final = [{"role": "user", "content": "what's the weather in SF"}]
chunks: Final = tuple(
completion(
model="gpt-6-luna",
reasoning_effort="none",
messages=messages,
functions=functions,
function_call={"name": "get_current_weather"},
stream=True,
max_tokens=128,
)
# Add any assertions here to check the response
print(response)
for chunk in response:
print(chunk)
if chunk["choices"][0]["finish_reason"] == "stop":
break
print(chunk["choices"][0]["finish_reason"])
print(chunk["choices"][0]["delta"]["content"])
except Exception as e:
pytest.fail(f"Error occurred: {e}")
)
response: Final = litellm.stream_chunk_builder(chunks, messages=messages)
assert response is not None
function_call: Final = response.choices[0].message.function_call
assert function_call is not None
assert function_call.name == "get_current_weather"
assert json.loads(function_call.arguments)["location"]
assert sum(chunk.choices[0].finish_reason is not None for chunk in chunks) == 1
#### Test Async streaming ####

View file

@ -1,4 +1,5 @@
import asyncio
from typing import Final
import json
import traceback
@ -3790,42 +3791,30 @@ def test_completion_openai_prompt():
# test_completion_openai_prompt()
def test_completion_openai_engine_and_model():
try:
print("\n text 003 test\n")
litellm.set_verbose = True
response = text_completion(
model="gpt-3.5-turbo-instruct",
engine="anything",
prompt="What's the weather in SF?",
max_tokens=5,
)
print(response)
response_str = response["choices"][0]["text"]
# print(response.choices[0])
# print(response.choices[0].text)
except Exception as e:
pytest.fail(f"Error occurred: {e}")
def test_completion_openai_engine_and_model() -> None:
response: Final = text_completion(
model="gpt-6-luna",
engine="anything",
reasoning_effort="none",
prompt="What's the weather in SF?",
max_tokens=5,
)
assert response.model == "gpt-6-luna"
assert response.choices[0].text
# test_completion_openai_engine_and_model()
def test_completion_openai_engine():
try:
print("\n text 003 test\n")
litellm.set_verbose = True
response = text_completion(
engine="gpt-3.5-turbo-instruct",
prompt="What's the weather in SF?",
max_tokens=5,
)
print(response)
response_str = response["choices"][0]["text"]
# print(response.choices[0])
# print(response.choices[0].text)
except Exception as e:
pytest.fail(f"Error occurred: {e}")
def test_completion_openai_engine() -> None:
response: Final = text_completion(
engine="gpt-6-luna",
reasoning_effort="none",
prompt="What's the weather in SF?",
max_tokens=5,
)
assert response.model == "gpt-6-luna"
assert response.choices[0].text
# test_completion_openai_engine()
@ -4048,34 +4037,18 @@ def test_async_text_completion_together_ai():
# test_async_text_completion()
def test_async_text_completion_stream():
# tests atext_completion + streaming - assert only one finish reason sent
litellm.set_verbose = False
print("test_async_text_completion with stream")
async def test_get_response():
try:
response = await litellm.atext_completion(
model="gpt-3.5-turbo-instruct",
prompt="good morning",
stream=True,
)
print(f"response: {response}")
num_finish_reason = 0
async for chunk in response:
print(chunk)
if chunk["choices"][0].get("finish_reason") is not None:
num_finish_reason += 1
print("finish_reason", chunk["choices"][0].get("finish_reason"))
assert (
num_finish_reason == 1
), f"expected only one finish reason. Got {num_finish_reason}"
except Exception as e:
pytest.fail(f"GOT exception for gpt-3.5 instruct In streaming{e}")
asyncio.run(test_get_response())
@pytest.mark.asyncio
async def test_async_text_completion_stream() -> None:
response: Final = await litellm.atext_completion(
model="gpt-6-luna",
reasoning_effort="none",
prompt="good morning",
stream=True,
max_tokens=32,
)
chunks: Final = [chunk async for chunk in response]
assert sum(chunk.choices[0].finish_reason is not None for chunk in chunks) == 1
assert any(chunk.choices[0].text for chunk in chunks)
# test_async_text_completion_stream()

View file

@ -1,3 +1,6 @@
import os
from typing import Final
# What is this?
## This tests if the proxy fallbacks work as expected
import pytest
@ -6,6 +9,9 @@ import aiohttp
from tests.large_text import text
import time
from typing import Optional
from openai import AsyncOpenAI, PermissionDeniedError
PROXY_BASE_URL: Final = os.environ.get("LITELLM_PROXY_BASE_URL", "http://0.0.0.0:4000")
async def generate_key(
@ -14,7 +20,7 @@ async def generate_key(
models: list,
calling_key="sk-1234",
):
url = "http://0.0.0.0:4000/key/generate"
url: Final = f"{PROXY_BASE_URL}/key/generate"
headers = {
"Authorization": f"Bearer {calling_key}",
"Content-Type": "application/json",
@ -48,7 +54,7 @@ async def chat_completion(
extra_headers: Optional[dict] = None,
**kwargs,
):
url = "http://0.0.0.0:4000/chat/completions"
url: Final = f"{PROXY_BASE_URL}/chat/completions"
headers = {
"Authorization": f"Bearer {key}",
"Content-Type": "application/json",
@ -94,42 +100,30 @@ async def test_chat_completion():
@pytest.mark.parametrize("has_access", [True, False])
@pytest.mark.asyncio
async def test_chat_completion_client_fallbacks(has_access):
"""
make chat completion call with prompt > context window. expect it to work with fallback
"""
async def test_chat_completion_client_fallbacks(has_access: bool) -> None:
models: Final = ["gpt-3.5-turbo", "gpt-6-luna"] if has_access else ["gpt-3.5-turbo"]
async with aiohttp.ClientSession() as session:
models = ["gpt-3.5-turbo"]
if has_access:
models.append("gpt-instruct")
## CREATE KEY WITH MODELS
generated_key = await generate_key(session=session, i=0, models=models)
calling_key = generated_key["key"]
model = "gpt-3.5-turbo"
messages = [
{"role": "user", "content": "Who was Alexander?"},
]
## CALL PROXY
try:
await chat_completion(
session=session,
key=calling_key,
model=model,
messages=messages,
mock_testing_fallbacks=True,
fallbacks=["gpt-instruct"],
)
if not has_access:
pytest.fail(
"Expected this to fail, submitted fallback model that key did not have access to"
)
except Exception as e:
if has_access:
pytest.fail("Expected this to work: {}".format(str(e)))
generated_key: Final = await generate_key(session=session, i=0, models=models)
async with AsyncOpenAI(api_key=generated_key["key"], base_url=PROXY_BASE_URL, max_retries=0) as client:
request: Final = {
"model": "gpt-3.5-turbo",
"messages": [{"role": "user", "content": "Who was Alexander?"}],
"max_tokens": 32,
"temperature": 0,
"extra_body": {
"mock_testing_fallbacks": True,
"fallbacks": ["gpt-6-luna"],
},
}
if not has_access:
with pytest.raises(PermissionDeniedError) as denied:
await client.chat.completions.create(**request)
assert denied.value.status_code == 403
assert "gpt-6-luna" in str(denied.value)
return
response: Final = await client.chat.completions.create(**request)
assert response.model == "gpt-6-luna"
assert response.choices[0].message.content
@pytest.mark.asyncio
@ -241,55 +235,66 @@ async def test_chat_completion_with_timeout_from_request():
@pytest.mark.parametrize("has_access", [True, False])
@pytest.mark.asyncio
async def test_chat_completion_client_fallbacks_with_custom_message(has_access):
"""
make chat completion call with prompt > context window. expect it to work with fallback
"""
async def test_chat_completion_client_fallbacks_with_custom_message(has_access: bool) -> None:
original_messages: Final = [{"role": "user", "content": "Who was Alexander?"}]
custom_messages: Final = [
{
"role": "user",
"content": (
"Describe the weather in a coastal city during winter, including the usual temperature, rain, wind, "
"and the clothing a visitor should bring."
),
}
]
models: Final = ["gpt-3.5-turbo", "gpt-6-luna"] if has_access else ["gpt-3.5-turbo"]
async with aiohttp.ClientSession() as session:
models = ["gpt-3.5-turbo"]
if has_access:
models.append("gpt-instruct")
## CREATE KEY WITH MODELS
generated_key = await generate_key(session=session, i=0, models=models)
calling_key = generated_key["key"]
model = "gpt-3.5-turbo"
messages = [
{"role": "user", "content": "Who was Alexander?"},
]
## CALL PROXY
try:
await chat_completion(
session=session,
key=calling_key,
model=model,
messages=messages,
mock_testing_fallbacks=True,
fallbacks=[
generated_key: Final = await generate_key(session=session, i=0, models=models)
async with AsyncOpenAI(api_key=generated_key["key"], base_url=PROXY_BASE_URL, max_retries=0) as client:
request: Final = {
"model": "gpt-3.5-turbo",
"messages": original_messages,
"max_tokens": 32,
"temperature": 0,
"extra_body": {
"mock_testing_fallbacks": True,
"fallbacks": [
{
"model": "gpt-instruct",
"messages": [
{
"role": "assistant",
"content": "This is a custom message",
}
],
"model": "gpt-6-luna",
"messages": custom_messages,
}
],
)
if not has_access:
pytest.fail(
"Expected this to fail, submitted fallback model that key did not have access to"
)
except Exception as e:
if has_access:
pytest.fail("Expected this to work: {}".format(str(e)))
},
}
if not has_access:
with pytest.raises(PermissionDeniedError) as denied:
await client.chat.completions.create(**request)
assert denied.value.status_code == 403
assert "gpt-6-luna" in str(denied.value)
return
response: Final = await client.chat.completions.create(**request)
assert response.model == "gpt-6-luna"
assert response.choices[0].message.content
custom_control: Final = await client.chat.completions.create(
model="gpt-6-luna",
messages=custom_messages,
max_tokens=32,
temperature=0,
)
original_control: Final = await client.chat.completions.create(
model="gpt-6-luna",
messages=original_messages,
max_tokens=32,
temperature=0,
)
assert response.usage is not None
assert custom_control.usage is not None
assert original_control.usage is not None
assert custom_control.usage.completion_tokens > 0
assert original_control.usage.completion_tokens > 0
assert custom_control.usage.prompt_tokens != original_control.usage.prompt_tokens
assert response.usage.prompt_tokens == custom_control.usage.prompt_tokens
from openai import AsyncOpenAI
from typing import List

View file

@ -1,3 +1,5 @@
import os
from typing import Final
# What this tests ?
## Tests /chat/completions by generating a key and then making a chat completions-request
import pytest
@ -398,10 +400,12 @@ async def test_completion_streaming_usage_metrics():
"""
[PROD Test] Ensures usage metrics are returned correctly when `include_usage` is set to `True`
"""
client = AsyncOpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000")
client: Final = AsyncOpenAI(
api_key="sk-1234", base_url=os.environ.get("LITELLM_PROXY_BASE_URL", "http://0.0.0.0:4000")
)
response = await client.completions.create(
model="gpt-instruct",
model="gpt-6-luna",
prompt="hey",
stream=True,
stream_options={"include_usage": True},
@ -417,9 +421,7 @@ async def test_completion_streaming_usage_metrics():
assert last_chunk is not None, "No chunks were received"
assert last_chunk.usage is not None, "Usage information was not received"
assert last_chunk.usage.prompt_tokens > 0, "Prompt tokens should be greater than 0"
assert (
last_chunk.usage.completion_tokens > 0
), "Completion tokens should be greater than 0"
assert last_chunk.usage.completion_tokens > 0, "Completion tokens should be greater than 0"
assert last_chunk.usage.total_tokens > 0, "Total tokens should be greater than 0"