mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
Add tests for openai evals
This commit is contained in:
parent
71385a4e12
commit
32263deb02
4 changed files with 497 additions and 1 deletions
|
|
@ -37,6 +37,7 @@ from litellm.llms.base_llm.batches.transformation import BaseBatchesConfig
|
|||
from litellm.llms.base_llm.chat.transformation import BaseConfig
|
||||
from litellm.llms.base_llm.containers.transformation import BaseContainerConfig
|
||||
from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig
|
||||
from litellm.llms.base_llm.evals.transformation import BaseEvalsAPIConfig
|
||||
from litellm.llms.base_llm.files.transformation import BaseFilesConfig
|
||||
from litellm.llms.base_llm.google_genai.transformation import (
|
||||
BaseGoogleGenAIGenerateContentConfig,
|
||||
|
|
@ -132,7 +133,6 @@ if TYPE_CHECKING:
|
|||
from aiohttp import ClientSession
|
||||
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
|
||||
from litellm.llms.base_llm.evals.transformation import BaseEvalsAPIConfig
|
||||
from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig
|
||||
from litellm.types.llms.openai_evals import (
|
||||
CancelEvalResponse,
|
||||
|
|
|
|||
237
tests/llm_translation/test_evals_api.py
Normal file
237
tests/llm_translation/test_evals_api.py
Normal file
|
|
@ -0,0 +1,237 @@
|
|||
"""
|
||||
Tests for Evals API operations across providers
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import Optional
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../.."))
|
||||
|
||||
import litellm
|
||||
from litellm.types.llms.openai_evals import (
|
||||
CancelEvalResponse,
|
||||
DeleteEvalResponse,
|
||||
Eval,
|
||||
ListEvalsResponse,
|
||||
)
|
||||
|
||||
|
||||
class BaseEvalsAPITest(ABC):
|
||||
"""
|
||||
Base test class for Evals API operations.
|
||||
Tests create, list, get, update, delete, and cancel operations.
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def get_custom_llm_provider(self) -> str:
|
||||
"""Return the provider name (e.g., 'openai')"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_api_key(self) -> Optional[str]:
|
||||
"""Return the API key for the provider"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_api_base(self) -> Optional[str]:
|
||||
"""Return the API base URL for the provider"""
|
||||
pass
|
||||
|
||||
def test_create_eval(self):
|
||||
"""
|
||||
Test creating an evaluation.
|
||||
"""
|
||||
import time
|
||||
|
||||
custom_llm_provider = self.get_custom_llm_provider()
|
||||
api_key = self.get_api_key()
|
||||
api_base = self.get_api_base()
|
||||
|
||||
if not api_key:
|
||||
pytest.skip(f"No API key provided for {custom_llm_provider}")
|
||||
|
||||
litellm.set_verbose = True
|
||||
|
||||
# Create eval with stored_completions data source
|
||||
unique_name = f"Test Eval {int(time.time())}"
|
||||
|
||||
response = litellm.create_eval(
|
||||
name=unique_name,
|
||||
data_source_config={
|
||||
"type": "stored_completions",
|
||||
"metadata": {"usecase": "chatbot"},
|
||||
},
|
||||
testing_criteria=[
|
||||
{
|
||||
"type": "label_model",
|
||||
"model": "gpt-4o",
|
||||
"input": [
|
||||
{
|
||||
"role": "developer",
|
||||
"content": "Classify the sentiment as 'positive' or 'negative'",
|
||||
},
|
||||
{"role": "user", "content": "Statement: {{item.input}}"},
|
||||
],
|
||||
"passing_labels": ["positive"],
|
||||
"labels": ["positive", "negative"],
|
||||
"name": "Sentiment grader",
|
||||
}
|
||||
],
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert response is not None
|
||||
assert isinstance(response, Eval)
|
||||
assert response.id is not None
|
||||
assert response.name == unique_name
|
||||
print(f"Created eval: {response}")
|
||||
print(f"Eval ID: {response.id}")
|
||||
|
||||
def test_list_evals(self):
|
||||
"""
|
||||
Test listing evaluations.
|
||||
"""
|
||||
custom_llm_provider = self.get_custom_llm_provider()
|
||||
api_key = self.get_api_key()
|
||||
api_base = self.get_api_base()
|
||||
|
||||
if not api_key:
|
||||
pytest.skip(f"No API key provided for {custom_llm_provider}")
|
||||
|
||||
litellm.set_verbose = True
|
||||
|
||||
response = litellm.list_evals(
|
||||
limit=10,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert response is not None
|
||||
assert isinstance(response, ListEvalsResponse)
|
||||
assert hasattr(response, "data")
|
||||
assert hasattr(response, "has_more")
|
||||
print(f"Listed evals: {len(response.data)} evaluations")
|
||||
|
||||
def test_get_eval(self):
|
||||
"""
|
||||
Test getting a specific evaluation by ID.
|
||||
"""
|
||||
custom_llm_provider = self.get_custom_llm_provider()
|
||||
api_key = self.get_api_key()
|
||||
api_base = self.get_api_base()
|
||||
|
||||
if not api_key:
|
||||
pytest.skip(f"No API key provided for {custom_llm_provider}")
|
||||
|
||||
litellm.set_verbose = True
|
||||
|
||||
# First list existing evals to get an ID
|
||||
list_response = litellm.list_evals(
|
||||
limit=1,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert isinstance(list_response, ListEvalsResponse)
|
||||
|
||||
if list_response.data and len(list_response.data) > 0:
|
||||
eval_id = list_response.data[0].id
|
||||
print(f"Testing with eval ID: {eval_id}")
|
||||
|
||||
# Get the eval
|
||||
response = litellm.get_eval(
|
||||
eval_id=eval_id,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert response is not None
|
||||
assert isinstance(response, Eval)
|
||||
assert response.id == eval_id
|
||||
print(f"Retrieved eval: {response}")
|
||||
else:
|
||||
pytest.skip("No existing evals to test with")
|
||||
|
||||
def test_update_eval(self):
|
||||
"""
|
||||
Test updating an evaluation.
|
||||
"""
|
||||
import time
|
||||
|
||||
custom_llm_provider = self.get_custom_llm_provider()
|
||||
api_key = self.get_api_key()
|
||||
api_base = self.get_api_base()
|
||||
|
||||
if not api_key:
|
||||
pytest.skip(f"No API key provided for {custom_llm_provider}")
|
||||
|
||||
litellm.set_verbose = True
|
||||
|
||||
# First list existing evals
|
||||
list_response = litellm.list_evals(
|
||||
limit=1,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert isinstance(list_response, ListEvalsResponse)
|
||||
|
||||
if list_response.data and len(list_response.data) > 0:
|
||||
eval_id = list_response.data[0].id
|
||||
updated_name = f"Updated Eval {int(time.time())}"
|
||||
|
||||
# Update the eval
|
||||
response = litellm.update_eval(
|
||||
eval_id=eval_id,
|
||||
name=updated_name,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert response is not None
|
||||
assert isinstance(response, Eval)
|
||||
assert response.id == eval_id
|
||||
assert response.name == updated_name
|
||||
print(f"Updated eval: {response}")
|
||||
else:
|
||||
pytest.skip("No existing evals to test with")
|
||||
|
||||
def test_delete_eval(self):
|
||||
"""
|
||||
Test deleting an evaluation.
|
||||
"""
|
||||
custom_llm_provider = self.get_custom_llm_provider()
|
||||
api_key = self.get_api_key()
|
||||
api_base = self.get_api_base()
|
||||
|
||||
if not api_key:
|
||||
pytest.skip(f"No API key provided for {custom_llm_provider}")
|
||||
|
||||
# Skip this test to avoid deleting production evals
|
||||
pytest.skip("Skipping delete test to preserve existing evals")
|
||||
|
||||
|
||||
class TestOpenAIEvalsAPI(BaseEvalsAPITest):
|
||||
"""
|
||||
Test OpenAI Evals API implementation.
|
||||
"""
|
||||
|
||||
def get_custom_llm_provider(self) -> str:
|
||||
return "openai"
|
||||
|
||||
def get_api_key(self) -> Optional[str]:
|
||||
return os.environ.get("OPENAI_API_KEY")
|
||||
|
||||
def get_api_base(self) -> Optional[str]:
|
||||
return os.environ.get("OPENAI_API_BASE")
|
||||
1
tests/test_litellm/llms/openai/evals/__init__.py
Normal file
1
tests/test_litellm/llms/openai/evals/__init__.py
Normal file
|
|
@ -0,0 +1 @@
|
|||
"""OpenAI Evals API tests"""
|
||||
|
|
@ -0,0 +1,258 @@
|
|||
"""
|
||||
Unit tests for OpenAI Evals API transformation
|
||||
"""
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from litellm.llms.openai.evals.transformation import OpenAIEvalsConfig
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def config() -> OpenAIEvalsConfig:
|
||||
return OpenAIEvalsConfig()
|
||||
|
||||
|
||||
def test_validate_environment_sets_headers(config: OpenAIEvalsConfig):
|
||||
"""Test that validate_environment correctly sets authorization headers"""
|
||||
headers: dict = {}
|
||||
params = GenericLiteLLMParams(api_key="sk-test-12345")
|
||||
|
||||
result = config.validate_environment(headers=headers, litellm_params=params)
|
||||
|
||||
assert result["Authorization"] == "Bearer sk-test-12345"
|
||||
assert result["Content-Type"] == "application/json"
|
||||
|
||||
|
||||
def test_validate_environment_requires_api_key(config: OpenAIEvalsConfig):
|
||||
"""Test that validate_environment raises error when no API key is provided"""
|
||||
import os
|
||||
|
||||
# Ensure OPENAI_API_KEY environment variable is None before validation
|
||||
if "OPENAI_API_KEY" in os.environ:
|
||||
del os.environ["OPENAI_API_KEY"]
|
||||
|
||||
headers: dict = {}
|
||||
params = GenericLiteLLMParams()
|
||||
|
||||
with pytest.raises(ValueError, match="OPENAI_API_KEY is required"):
|
||||
config.validate_environment(headers=headers, litellm_params=params)
|
||||
|
||||
|
||||
def test_get_complete_url_with_eval_id(config: OpenAIEvalsConfig):
|
||||
"""Test URL construction with eval_id"""
|
||||
url = config.get_complete_url(
|
||||
api_base="https://api.openai.com",
|
||||
endpoint="evals",
|
||||
eval_id="eval_123",
|
||||
)
|
||||
assert url == "https://api.openai.com/v1/evals/eval_123"
|
||||
|
||||
|
||||
def test_get_complete_url_without_eval_id(config: OpenAIEvalsConfig):
|
||||
"""Test URL construction without eval_id"""
|
||||
url = config.get_complete_url(
|
||||
api_base="https://api.openai.com",
|
||||
endpoint="evals",
|
||||
)
|
||||
assert url == "https://api.openai.com/v1/evals"
|
||||
|
||||
|
||||
def test_transform_create_eval_request(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of create eval request"""
|
||||
create_request = {
|
||||
"name": "Test Eval",
|
||||
"data_source_config": {
|
||||
"type": "stored_completions",
|
||||
"metadata": {"usecase": "chatbot"}
|
||||
},
|
||||
"testing_criteria": [
|
||||
{
|
||||
"type": "label_model",
|
||||
"model": "gpt-4o",
|
||||
"input": [{"role": "user", "content": "Test"}],
|
||||
"passing_labels": ["positive"],
|
||||
"labels": ["positive", "negative"],
|
||||
"name": "Test Grader"
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
result = config.transform_create_eval_request(
|
||||
create_request=create_request,
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["name"] == "Test Eval"
|
||||
assert result["data_source_config"]["type"] == "stored_completions"
|
||||
assert len(result["testing_criteria"]) == 1
|
||||
assert result["testing_criteria"][0]["type"] == "label_model"
|
||||
|
||||
|
||||
def test_transform_create_eval_response(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of create eval response"""
|
||||
response = httpx.Response(
|
||||
status_code=200,
|
||||
json={
|
||||
"id": "eval_123",
|
||||
"object": "eval",
|
||||
"created_at": 1234567890,
|
||||
"name": "Test Eval",
|
||||
"data_source_config": {"type": "stored_completions"},
|
||||
"testing_criteria": [],
|
||||
},
|
||||
request=httpx.Request("POST", "https://api.openai.com/v1/evals"),
|
||||
)
|
||||
|
||||
result = config.transform_create_eval_response(
|
||||
raw_response=response,
|
||||
logging_obj=None, # type: ignore
|
||||
)
|
||||
|
||||
assert result.id == "eval_123"
|
||||
assert result.object == "eval"
|
||||
assert result.name == "Test Eval"
|
||||
|
||||
|
||||
def test_transform_list_evals_request(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of list evals request"""
|
||||
list_params = {
|
||||
"limit": 10,
|
||||
"after": "eval_123",
|
||||
"order": "desc",
|
||||
}
|
||||
|
||||
url, query_params = config.transform_list_evals_request(
|
||||
list_params=list_params,
|
||||
litellm_params=GenericLiteLLMParams(api_base="https://api.openai.com"),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert url == "https://api.openai.com/v1/evals"
|
||||
assert query_params["limit"] == 10
|
||||
assert query_params["after"] == "eval_123"
|
||||
assert query_params["order"] == "desc"
|
||||
|
||||
|
||||
def test_transform_list_evals_response(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of list evals response"""
|
||||
response = httpx.Response(
|
||||
status_code=200,
|
||||
json={
|
||||
"object": "list",
|
||||
"data": [
|
||||
{
|
||||
"id": "eval_123",
|
||||
"object": "eval",
|
||||
"created_at": 1234567890,
|
||||
"name": "Test Eval",
|
||||
"data_source_config": {"type": "stored_completions"},
|
||||
"testing_criteria": [],
|
||||
}
|
||||
],
|
||||
"first_id": "eval_123",
|
||||
"last_id": "eval_123",
|
||||
"has_more": False,
|
||||
},
|
||||
request=httpx.Request("GET", "https://api.openai.com/v1/evals"),
|
||||
)
|
||||
|
||||
result = config.transform_list_evals_response(
|
||||
raw_response=response,
|
||||
logging_obj=None, # type: ignore
|
||||
)
|
||||
|
||||
assert result.object == "list"
|
||||
assert len(result.data) == 1
|
||||
assert result.data[0].id == "eval_123"
|
||||
assert result.has_more is False
|
||||
|
||||
|
||||
def test_transform_update_eval_request(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of update eval request"""
|
||||
update_request = {
|
||||
"name": "Updated Eval Name",
|
||||
"metadata": {"key": "value"},
|
||||
}
|
||||
|
||||
url, headers, request_body = config.transform_update_eval_request(
|
||||
eval_id="eval_123",
|
||||
update_request=update_request,
|
||||
api_base="https://api.openai.com",
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert url == "https://api.openai.com/v1/evals/eval_123"
|
||||
assert request_body["name"] == "Updated Eval Name"
|
||||
assert request_body["metadata"]["key"] == "value"
|
||||
|
||||
|
||||
def test_transform_delete_eval_request(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of delete eval request"""
|
||||
url, headers = config.transform_delete_eval_request(
|
||||
eval_id="eval_123",
|
||||
api_base="https://api.openai.com",
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert url == "https://api.openai.com/v1/evals/eval_123"
|
||||
|
||||
|
||||
def test_transform_delete_eval_response(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of delete eval response"""
|
||||
response = httpx.Response(
|
||||
status_code=200,
|
||||
json={
|
||||
"object": "eval.deleted",
|
||||
"deleted": True,
|
||||
"eval_id": "eval_abc123"
|
||||
},
|
||||
request=httpx.Request("DELETE", "https://api.openai.com/v1/evals/eval_123"),
|
||||
)
|
||||
|
||||
result = config.transform_delete_eval_response(
|
||||
raw_response=response,
|
||||
logging_obj=None, # type: ignore
|
||||
)
|
||||
|
||||
assert result.eval_id == "eval_abc123"
|
||||
assert result.object == "eval.deleted"
|
||||
assert result.deleted is True
|
||||
|
||||
|
||||
def test_transform_cancel_eval_request(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of cancel eval request"""
|
||||
url, headers, request_body = config.transform_cancel_eval_request(
|
||||
eval_id="eval_123",
|
||||
api_base="https://api.openai.com",
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert url == "https://api.openai.com/v1/evals/eval_123/cancel"
|
||||
assert request_body == {}
|
||||
|
||||
|
||||
def test_transform_cancel_eval_response(config: OpenAIEvalsConfig):
|
||||
"""Test transformation of cancel eval response"""
|
||||
response = httpx.Response(
|
||||
status_code=200,
|
||||
json={
|
||||
"id": "eval_123",
|
||||
"object": "eval",
|
||||
"status": "cancelled",
|
||||
},
|
||||
request=httpx.Request("POST", "https://api.openai.com/v1/evals/eval_123/cancel"),
|
||||
)
|
||||
|
||||
result = config.transform_cancel_eval_response(
|
||||
raw_response=response,
|
||||
logging_obj=None, # type: ignore
|
||||
)
|
||||
|
||||
assert result.id == "eval_123"
|
||||
assert result.object == "eval"
|
||||
Loading…
Add table
Reference in a new issue