Add tests for openai evals

This commit is contained in:
Sameer Kankute 2026-02-17 17:52:20 +05:30
parent 71385a4e12
commit 32263deb02
4 changed files with 497 additions and 1 deletions

View file

@ -37,6 +37,7 @@ from litellm.llms.base_llm.batches.transformation import BaseBatchesConfig
from litellm.llms.base_llm.chat.transformation import BaseConfig
from litellm.llms.base_llm.containers.transformation import BaseContainerConfig
from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig
from litellm.llms.base_llm.evals.transformation import BaseEvalsAPIConfig
from litellm.llms.base_llm.files.transformation import BaseFilesConfig
from litellm.llms.base_llm.google_genai.transformation import (
BaseGoogleGenAIGenerateContentConfig,
@ -132,7 +133,6 @@ if TYPE_CHECKING:
from aiohttp import ClientSession
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
from litellm.llms.base_llm.evals.transformation import BaseEvalsAPIConfig
from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig
from litellm.types.llms.openai_evals import (
CancelEvalResponse,

View file

@ -0,0 +1,237 @@
"""
Tests for Evals API operations across providers
"""
import os
import sys
from abc import ABC, abstractmethod
from typing import Optional
import pytest
sys.path.insert(0, os.path.abspath("../.."))
import litellm
from litellm.types.llms.openai_evals import (
CancelEvalResponse,
DeleteEvalResponse,
Eval,
ListEvalsResponse,
)
class BaseEvalsAPITest(ABC):
"""
Base test class for Evals API operations.
Tests create, list, get, update, delete, and cancel operations.
"""
@abstractmethod
def get_custom_llm_provider(self) -> str:
"""Return the provider name (e.g., 'openai')"""
pass
@abstractmethod
def get_api_key(self) -> Optional[str]:
"""Return the API key for the provider"""
pass
@abstractmethod
def get_api_base(self) -> Optional[str]:
"""Return the API base URL for the provider"""
pass
def test_create_eval(self):
"""
Test creating an evaluation.
"""
import time
custom_llm_provider = self.get_custom_llm_provider()
api_key = self.get_api_key()
api_base = self.get_api_base()
if not api_key:
pytest.skip(f"No API key provided for {custom_llm_provider}")
litellm.set_verbose = True
# Create eval with stored_completions data source
unique_name = f"Test Eval {int(time.time())}"
response = litellm.create_eval(
name=unique_name,
data_source_config={
"type": "stored_completions",
"metadata": {"usecase": "chatbot"},
},
testing_criteria=[
{
"type": "label_model",
"model": "gpt-4o",
"input": [
{
"role": "developer",
"content": "Classify the sentiment as 'positive' or 'negative'",
},
{"role": "user", "content": "Statement: {{item.input}}"},
],
"passing_labels": ["positive"],
"labels": ["positive", "negative"],
"name": "Sentiment grader",
}
],
custom_llm_provider=custom_llm_provider,
api_key=api_key,
api_base=api_base,
)
assert response is not None
assert isinstance(response, Eval)
assert response.id is not None
assert response.name == unique_name
print(f"Created eval: {response}")
print(f"Eval ID: {response.id}")
def test_list_evals(self):
"""
Test listing evaluations.
"""
custom_llm_provider = self.get_custom_llm_provider()
api_key = self.get_api_key()
api_base = self.get_api_base()
if not api_key:
pytest.skip(f"No API key provided for {custom_llm_provider}")
litellm.set_verbose = True
response = litellm.list_evals(
limit=10,
custom_llm_provider=custom_llm_provider,
api_key=api_key,
api_base=api_base,
)
assert response is not None
assert isinstance(response, ListEvalsResponse)
assert hasattr(response, "data")
assert hasattr(response, "has_more")
print(f"Listed evals: {len(response.data)} evaluations")
def test_get_eval(self):
"""
Test getting a specific evaluation by ID.
"""
custom_llm_provider = self.get_custom_llm_provider()
api_key = self.get_api_key()
api_base = self.get_api_base()
if not api_key:
pytest.skip(f"No API key provided for {custom_llm_provider}")
litellm.set_verbose = True
# First list existing evals to get an ID
list_response = litellm.list_evals(
limit=1,
custom_llm_provider=custom_llm_provider,
api_key=api_key,
api_base=api_base,
)
assert isinstance(list_response, ListEvalsResponse)
if list_response.data and len(list_response.data) > 0:
eval_id = list_response.data[0].id
print(f"Testing with eval ID: {eval_id}")
# Get the eval
response = litellm.get_eval(
eval_id=eval_id,
custom_llm_provider=custom_llm_provider,
api_key=api_key,
api_base=api_base,
)
assert response is not None
assert isinstance(response, Eval)
assert response.id == eval_id
print(f"Retrieved eval: {response}")
else:
pytest.skip("No existing evals to test with")
def test_update_eval(self):
"""
Test updating an evaluation.
"""
import time
custom_llm_provider = self.get_custom_llm_provider()
api_key = self.get_api_key()
api_base = self.get_api_base()
if not api_key:
pytest.skip(f"No API key provided for {custom_llm_provider}")
litellm.set_verbose = True
# First list existing evals
list_response = litellm.list_evals(
limit=1,
custom_llm_provider=custom_llm_provider,
api_key=api_key,
api_base=api_base,
)
assert isinstance(list_response, ListEvalsResponse)
if list_response.data and len(list_response.data) > 0:
eval_id = list_response.data[0].id
updated_name = f"Updated Eval {int(time.time())}"
# Update the eval
response = litellm.update_eval(
eval_id=eval_id,
name=updated_name,
custom_llm_provider=custom_llm_provider,
api_key=api_key,
api_base=api_base,
)
assert response is not None
assert isinstance(response, Eval)
assert response.id == eval_id
assert response.name == updated_name
print(f"Updated eval: {response}")
else:
pytest.skip("No existing evals to test with")
def test_delete_eval(self):
"""
Test deleting an evaluation.
"""
custom_llm_provider = self.get_custom_llm_provider()
api_key = self.get_api_key()
api_base = self.get_api_base()
if not api_key:
pytest.skip(f"No API key provided for {custom_llm_provider}")
# Skip this test to avoid deleting production evals
pytest.skip("Skipping delete test to preserve existing evals")
class TestOpenAIEvalsAPI(BaseEvalsAPITest):
"""
Test OpenAI Evals API implementation.
"""
def get_custom_llm_provider(self) -> str:
return "openai"
def get_api_key(self) -> Optional[str]:
return os.environ.get("OPENAI_API_KEY")
def get_api_base(self) -> Optional[str]:
return os.environ.get("OPENAI_API_BASE")

View file

@ -0,0 +1 @@
"""OpenAI Evals API tests"""

View file

@ -0,0 +1,258 @@
"""
Unit tests for OpenAI Evals API transformation
"""
import httpx
import pytest
from litellm.llms.openai.evals.transformation import OpenAIEvalsConfig
from litellm.types.router import GenericLiteLLMParams
@pytest.fixture()
def config() -> OpenAIEvalsConfig:
return OpenAIEvalsConfig()
def test_validate_environment_sets_headers(config: OpenAIEvalsConfig):
"""Test that validate_environment correctly sets authorization headers"""
headers: dict = {}
params = GenericLiteLLMParams(api_key="sk-test-12345")
result = config.validate_environment(headers=headers, litellm_params=params)
assert result["Authorization"] == "Bearer sk-test-12345"
assert result["Content-Type"] == "application/json"
def test_validate_environment_requires_api_key(config: OpenAIEvalsConfig):
"""Test that validate_environment raises error when no API key is provided"""
import os
# Ensure OPENAI_API_KEY environment variable is None before validation
if "OPENAI_API_KEY" in os.environ:
del os.environ["OPENAI_API_KEY"]
headers: dict = {}
params = GenericLiteLLMParams()
with pytest.raises(ValueError, match="OPENAI_API_KEY is required"):
config.validate_environment(headers=headers, litellm_params=params)
def test_get_complete_url_with_eval_id(config: OpenAIEvalsConfig):
"""Test URL construction with eval_id"""
url = config.get_complete_url(
api_base="https://api.openai.com",
endpoint="evals",
eval_id="eval_123",
)
assert url == "https://api.openai.com/v1/evals/eval_123"
def test_get_complete_url_without_eval_id(config: OpenAIEvalsConfig):
"""Test URL construction without eval_id"""
url = config.get_complete_url(
api_base="https://api.openai.com",
endpoint="evals",
)
assert url == "https://api.openai.com/v1/evals"
def test_transform_create_eval_request(config: OpenAIEvalsConfig):
"""Test transformation of create eval request"""
create_request = {
"name": "Test Eval",
"data_source_config": {
"type": "stored_completions",
"metadata": {"usecase": "chatbot"}
},
"testing_criteria": [
{
"type": "label_model",
"model": "gpt-4o",
"input": [{"role": "user", "content": "Test"}],
"passing_labels": ["positive"],
"labels": ["positive", "negative"],
"name": "Test Grader"
}
],
}
result = config.transform_create_eval_request(
create_request=create_request,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert result["name"] == "Test Eval"
assert result["data_source_config"]["type"] == "stored_completions"
assert len(result["testing_criteria"]) == 1
assert result["testing_criteria"][0]["type"] == "label_model"
def test_transform_create_eval_response(config: OpenAIEvalsConfig):
"""Test transformation of create eval response"""
response = httpx.Response(
status_code=200,
json={
"id": "eval_123",
"object": "eval",
"created_at": 1234567890,
"name": "Test Eval",
"data_source_config": {"type": "stored_completions"},
"testing_criteria": [],
},
request=httpx.Request("POST", "https://api.openai.com/v1/evals"),
)
result = config.transform_create_eval_response(
raw_response=response,
logging_obj=None, # type: ignore
)
assert result.id == "eval_123"
assert result.object == "eval"
assert result.name == "Test Eval"
def test_transform_list_evals_request(config: OpenAIEvalsConfig):
"""Test transformation of list evals request"""
list_params = {
"limit": 10,
"after": "eval_123",
"order": "desc",
}
url, query_params = config.transform_list_evals_request(
list_params=list_params,
litellm_params=GenericLiteLLMParams(api_base="https://api.openai.com"),
headers={},
)
assert url == "https://api.openai.com/v1/evals"
assert query_params["limit"] == 10
assert query_params["after"] == "eval_123"
assert query_params["order"] == "desc"
def test_transform_list_evals_response(config: OpenAIEvalsConfig):
"""Test transformation of list evals response"""
response = httpx.Response(
status_code=200,
json={
"object": "list",
"data": [
{
"id": "eval_123",
"object": "eval",
"created_at": 1234567890,
"name": "Test Eval",
"data_source_config": {"type": "stored_completions"},
"testing_criteria": [],
}
],
"first_id": "eval_123",
"last_id": "eval_123",
"has_more": False,
},
request=httpx.Request("GET", "https://api.openai.com/v1/evals"),
)
result = config.transform_list_evals_response(
raw_response=response,
logging_obj=None, # type: ignore
)
assert result.object == "list"
assert len(result.data) == 1
assert result.data[0].id == "eval_123"
assert result.has_more is False
def test_transform_update_eval_request(config: OpenAIEvalsConfig):
"""Test transformation of update eval request"""
update_request = {
"name": "Updated Eval Name",
"metadata": {"key": "value"},
}
url, headers, request_body = config.transform_update_eval_request(
eval_id="eval_123",
update_request=update_request,
api_base="https://api.openai.com",
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert url == "https://api.openai.com/v1/evals/eval_123"
assert request_body["name"] == "Updated Eval Name"
assert request_body["metadata"]["key"] == "value"
def test_transform_delete_eval_request(config: OpenAIEvalsConfig):
"""Test transformation of delete eval request"""
url, headers = config.transform_delete_eval_request(
eval_id="eval_123",
api_base="https://api.openai.com",
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert url == "https://api.openai.com/v1/evals/eval_123"
def test_transform_delete_eval_response(config: OpenAIEvalsConfig):
"""Test transformation of delete eval response"""
response = httpx.Response(
status_code=200,
json={
"object": "eval.deleted",
"deleted": True,
"eval_id": "eval_abc123"
},
request=httpx.Request("DELETE", "https://api.openai.com/v1/evals/eval_123"),
)
result = config.transform_delete_eval_response(
raw_response=response,
logging_obj=None, # type: ignore
)
assert result.eval_id == "eval_abc123"
assert result.object == "eval.deleted"
assert result.deleted is True
def test_transform_cancel_eval_request(config: OpenAIEvalsConfig):
"""Test transformation of cancel eval request"""
url, headers, request_body = config.transform_cancel_eval_request(
eval_id="eval_123",
api_base="https://api.openai.com",
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert url == "https://api.openai.com/v1/evals/eval_123/cancel"
assert request_body == {}
def test_transform_cancel_eval_response(config: OpenAIEvalsConfig):
"""Test transformation of cancel eval response"""
response = httpx.Response(
status_code=200,
json={
"id": "eval_123",
"object": "eval",
"status": "cancelled",
},
request=httpx.Request("POST", "https://api.openai.com/v1/evals/eval_123/cancel"),
)
result = config.transform_cancel_eval_response(
raw_response=response,
logging_obj=None, # type: ignore
)
assert result.id == "eval_123"
assert result.object == "eval"