From 32263deb02d998911dae5dc06612c3915db8a20f Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 17 Feb 2026 17:52:20 +0530 Subject: [PATCH] Add tests for openai evals --- litellm/llms/custom_httpx/llm_http_handler.py | 2 +- tests/llm_translation/test_evals_api.py | 237 ++++++++++++++++ .../llms/openai/evals/__init__.py | 1 + .../evals/test_openai_evals_transformation.py | 258 ++++++++++++++++++ 4 files changed, 497 insertions(+), 1 deletion(-) create mode 100644 tests/llm_translation/test_evals_api.py create mode 100644 tests/test_litellm/llms/openai/evals/__init__.py create mode 100644 tests/test_litellm/llms/openai/evals/test_openai_evals_transformation.py diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 366adf0978f..8f7345bae64 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -37,6 +37,7 @@ from litellm.llms.base_llm.batches.transformation import BaseBatchesConfig from litellm.llms.base_llm.chat.transformation import BaseConfig from litellm.llms.base_llm.containers.transformation import BaseContainerConfig from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig +from litellm.llms.base_llm.evals.transformation import BaseEvalsAPIConfig from litellm.llms.base_llm.files.transformation import BaseFilesConfig from litellm.llms.base_llm.google_genai.transformation import ( BaseGoogleGenAIGenerateContentConfig, @@ -132,7 +133,6 @@ if TYPE_CHECKING: from aiohttp import ClientSession from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj - from litellm.llms.base_llm.evals.transformation import BaseEvalsAPIConfig from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig from litellm.types.llms.openai_evals import ( CancelEvalResponse, diff --git a/tests/llm_translation/test_evals_api.py b/tests/llm_translation/test_evals_api.py new file mode 100644 index 00000000000..cc8f7897cd0 --- /dev/null +++ b/tests/llm_translation/test_evals_api.py @@ -0,0 +1,237 @@ +""" +Tests for Evals API operations across providers +""" + +import os +import sys +from abc import ABC, abstractmethod +from typing import Optional + +import pytest + +sys.path.insert(0, os.path.abspath("../..")) + +import litellm +from litellm.types.llms.openai_evals import ( + CancelEvalResponse, + DeleteEvalResponse, + Eval, + ListEvalsResponse, +) + + +class BaseEvalsAPITest(ABC): + """ + Base test class for Evals API operations. + Tests create, list, get, update, delete, and cancel operations. + """ + + @abstractmethod + def get_custom_llm_provider(self) -> str: + """Return the provider name (e.g., 'openai')""" + pass + + @abstractmethod + def get_api_key(self) -> Optional[str]: + """Return the API key for the provider""" + pass + + @abstractmethod + def get_api_base(self) -> Optional[str]: + """Return the API base URL for the provider""" + pass + + def test_create_eval(self): + """ + Test creating an evaluation. + """ + import time + + custom_llm_provider = self.get_custom_llm_provider() + api_key = self.get_api_key() + api_base = self.get_api_base() + + if not api_key: + pytest.skip(f"No API key provided for {custom_llm_provider}") + + litellm.set_verbose = True + + # Create eval with stored_completions data source + unique_name = f"Test Eval {int(time.time())}" + + response = litellm.create_eval( + name=unique_name, + data_source_config={ + "type": "stored_completions", + "metadata": {"usecase": "chatbot"}, + }, + testing_criteria=[ + { + "type": "label_model", + "model": "gpt-4o", + "input": [ + { + "role": "developer", + "content": "Classify the sentiment as 'positive' or 'negative'", + }, + {"role": "user", "content": "Statement: {{item.input}}"}, + ], + "passing_labels": ["positive"], + "labels": ["positive", "negative"], + "name": "Sentiment grader", + } + ], + custom_llm_provider=custom_llm_provider, + api_key=api_key, + api_base=api_base, + ) + + assert response is not None + assert isinstance(response, Eval) + assert response.id is not None + assert response.name == unique_name + print(f"Created eval: {response}") + print(f"Eval ID: {response.id}") + + def test_list_evals(self): + """ + Test listing evaluations. + """ + custom_llm_provider = self.get_custom_llm_provider() + api_key = self.get_api_key() + api_base = self.get_api_base() + + if not api_key: + pytest.skip(f"No API key provided for {custom_llm_provider}") + + litellm.set_verbose = True + + response = litellm.list_evals( + limit=10, + custom_llm_provider=custom_llm_provider, + api_key=api_key, + api_base=api_base, + ) + + assert response is not None + assert isinstance(response, ListEvalsResponse) + assert hasattr(response, "data") + assert hasattr(response, "has_more") + print(f"Listed evals: {len(response.data)} evaluations") + + def test_get_eval(self): + """ + Test getting a specific evaluation by ID. + """ + custom_llm_provider = self.get_custom_llm_provider() + api_key = self.get_api_key() + api_base = self.get_api_base() + + if not api_key: + pytest.skip(f"No API key provided for {custom_llm_provider}") + + litellm.set_verbose = True + + # First list existing evals to get an ID + list_response = litellm.list_evals( + limit=1, + custom_llm_provider=custom_llm_provider, + api_key=api_key, + api_base=api_base, + ) + + assert isinstance(list_response, ListEvalsResponse) + + if list_response.data and len(list_response.data) > 0: + eval_id = list_response.data[0].id + print(f"Testing with eval ID: {eval_id}") + + # Get the eval + response = litellm.get_eval( + eval_id=eval_id, + custom_llm_provider=custom_llm_provider, + api_key=api_key, + api_base=api_base, + ) + + assert response is not None + assert isinstance(response, Eval) + assert response.id == eval_id + print(f"Retrieved eval: {response}") + else: + pytest.skip("No existing evals to test with") + + def test_update_eval(self): + """ + Test updating an evaluation. + """ + import time + + custom_llm_provider = self.get_custom_llm_provider() + api_key = self.get_api_key() + api_base = self.get_api_base() + + if not api_key: + pytest.skip(f"No API key provided for {custom_llm_provider}") + + litellm.set_verbose = True + + # First list existing evals + list_response = litellm.list_evals( + limit=1, + custom_llm_provider=custom_llm_provider, + api_key=api_key, + api_base=api_base, + ) + + assert isinstance(list_response, ListEvalsResponse) + + if list_response.data and len(list_response.data) > 0: + eval_id = list_response.data[0].id + updated_name = f"Updated Eval {int(time.time())}" + + # Update the eval + response = litellm.update_eval( + eval_id=eval_id, + name=updated_name, + custom_llm_provider=custom_llm_provider, + api_key=api_key, + api_base=api_base, + ) + + assert response is not None + assert isinstance(response, Eval) + assert response.id == eval_id + assert response.name == updated_name + print(f"Updated eval: {response}") + else: + pytest.skip("No existing evals to test with") + + def test_delete_eval(self): + """ + Test deleting an evaluation. + """ + custom_llm_provider = self.get_custom_llm_provider() + api_key = self.get_api_key() + api_base = self.get_api_base() + + if not api_key: + pytest.skip(f"No API key provided for {custom_llm_provider}") + + # Skip this test to avoid deleting production evals + pytest.skip("Skipping delete test to preserve existing evals") + + +class TestOpenAIEvalsAPI(BaseEvalsAPITest): + """ + Test OpenAI Evals API implementation. + """ + + def get_custom_llm_provider(self) -> str: + return "openai" + + def get_api_key(self) -> Optional[str]: + return os.environ.get("OPENAI_API_KEY") + + def get_api_base(self) -> Optional[str]: + return os.environ.get("OPENAI_API_BASE") diff --git a/tests/test_litellm/llms/openai/evals/__init__.py b/tests/test_litellm/llms/openai/evals/__init__.py new file mode 100644 index 00000000000..47a8a2f0aed --- /dev/null +++ b/tests/test_litellm/llms/openai/evals/__init__.py @@ -0,0 +1 @@ +"""OpenAI Evals API tests""" diff --git a/tests/test_litellm/llms/openai/evals/test_openai_evals_transformation.py b/tests/test_litellm/llms/openai/evals/test_openai_evals_transformation.py new file mode 100644 index 00000000000..b70ef371957 --- /dev/null +++ b/tests/test_litellm/llms/openai/evals/test_openai_evals_transformation.py @@ -0,0 +1,258 @@ +""" +Unit tests for OpenAI Evals API transformation +""" + +import httpx +import pytest + +from litellm.llms.openai.evals.transformation import OpenAIEvalsConfig +from litellm.types.router import GenericLiteLLMParams + + +@pytest.fixture() +def config() -> OpenAIEvalsConfig: + return OpenAIEvalsConfig() + + +def test_validate_environment_sets_headers(config: OpenAIEvalsConfig): + """Test that validate_environment correctly sets authorization headers""" + headers: dict = {} + params = GenericLiteLLMParams(api_key="sk-test-12345") + + result = config.validate_environment(headers=headers, litellm_params=params) + + assert result["Authorization"] == "Bearer sk-test-12345" + assert result["Content-Type"] == "application/json" + + +def test_validate_environment_requires_api_key(config: OpenAIEvalsConfig): + """Test that validate_environment raises error when no API key is provided""" + import os + + # Ensure OPENAI_API_KEY environment variable is None before validation + if "OPENAI_API_KEY" in os.environ: + del os.environ["OPENAI_API_KEY"] + + headers: dict = {} + params = GenericLiteLLMParams() + + with pytest.raises(ValueError, match="OPENAI_API_KEY is required"): + config.validate_environment(headers=headers, litellm_params=params) + + +def test_get_complete_url_with_eval_id(config: OpenAIEvalsConfig): + """Test URL construction with eval_id""" + url = config.get_complete_url( + api_base="https://api.openai.com", + endpoint="evals", + eval_id="eval_123", + ) + assert url == "https://api.openai.com/v1/evals/eval_123" + + +def test_get_complete_url_without_eval_id(config: OpenAIEvalsConfig): + """Test URL construction without eval_id""" + url = config.get_complete_url( + api_base="https://api.openai.com", + endpoint="evals", + ) + assert url == "https://api.openai.com/v1/evals" + + +def test_transform_create_eval_request(config: OpenAIEvalsConfig): + """Test transformation of create eval request""" + create_request = { + "name": "Test Eval", + "data_source_config": { + "type": "stored_completions", + "metadata": {"usecase": "chatbot"} + }, + "testing_criteria": [ + { + "type": "label_model", + "model": "gpt-4o", + "input": [{"role": "user", "content": "Test"}], + "passing_labels": ["positive"], + "labels": ["positive", "negative"], + "name": "Test Grader" + } + ], + } + + result = config.transform_create_eval_request( + create_request=create_request, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert result["name"] == "Test Eval" + assert result["data_source_config"]["type"] == "stored_completions" + assert len(result["testing_criteria"]) == 1 + assert result["testing_criteria"][0]["type"] == "label_model" + + +def test_transform_create_eval_response(config: OpenAIEvalsConfig): + """Test transformation of create eval response""" + response = httpx.Response( + status_code=200, + json={ + "id": "eval_123", + "object": "eval", + "created_at": 1234567890, + "name": "Test Eval", + "data_source_config": {"type": "stored_completions"}, + "testing_criteria": [], + }, + request=httpx.Request("POST", "https://api.openai.com/v1/evals"), + ) + + result = config.transform_create_eval_response( + raw_response=response, + logging_obj=None, # type: ignore + ) + + assert result.id == "eval_123" + assert result.object == "eval" + assert result.name == "Test Eval" + + +def test_transform_list_evals_request(config: OpenAIEvalsConfig): + """Test transformation of list evals request""" + list_params = { + "limit": 10, + "after": "eval_123", + "order": "desc", + } + + url, query_params = config.transform_list_evals_request( + list_params=list_params, + litellm_params=GenericLiteLLMParams(api_base="https://api.openai.com"), + headers={}, + ) + + assert url == "https://api.openai.com/v1/evals" + assert query_params["limit"] == 10 + assert query_params["after"] == "eval_123" + assert query_params["order"] == "desc" + + +def test_transform_list_evals_response(config: OpenAIEvalsConfig): + """Test transformation of list evals response""" + response = httpx.Response( + status_code=200, + json={ + "object": "list", + "data": [ + { + "id": "eval_123", + "object": "eval", + "created_at": 1234567890, + "name": "Test Eval", + "data_source_config": {"type": "stored_completions"}, + "testing_criteria": [], + } + ], + "first_id": "eval_123", + "last_id": "eval_123", + "has_more": False, + }, + request=httpx.Request("GET", "https://api.openai.com/v1/evals"), + ) + + result = config.transform_list_evals_response( + raw_response=response, + logging_obj=None, # type: ignore + ) + + assert result.object == "list" + assert len(result.data) == 1 + assert result.data[0].id == "eval_123" + assert result.has_more is False + + +def test_transform_update_eval_request(config: OpenAIEvalsConfig): + """Test transformation of update eval request""" + update_request = { + "name": "Updated Eval Name", + "metadata": {"key": "value"}, + } + + url, headers, request_body = config.transform_update_eval_request( + eval_id="eval_123", + update_request=update_request, + api_base="https://api.openai.com", + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == "https://api.openai.com/v1/evals/eval_123" + assert request_body["name"] == "Updated Eval Name" + assert request_body["metadata"]["key"] == "value" + + +def test_transform_delete_eval_request(config: OpenAIEvalsConfig): + """Test transformation of delete eval request""" + url, headers = config.transform_delete_eval_request( + eval_id="eval_123", + api_base="https://api.openai.com", + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == "https://api.openai.com/v1/evals/eval_123" + + +def test_transform_delete_eval_response(config: OpenAIEvalsConfig): + """Test transformation of delete eval response""" + response = httpx.Response( + status_code=200, + json={ + "object": "eval.deleted", + "deleted": True, + "eval_id": "eval_abc123" + }, + request=httpx.Request("DELETE", "https://api.openai.com/v1/evals/eval_123"), + ) + + result = config.transform_delete_eval_response( + raw_response=response, + logging_obj=None, # type: ignore + ) + + assert result.eval_id == "eval_abc123" + assert result.object == "eval.deleted" + assert result.deleted is True + + +def test_transform_cancel_eval_request(config: OpenAIEvalsConfig): + """Test transformation of cancel eval request""" + url, headers, request_body = config.transform_cancel_eval_request( + eval_id="eval_123", + api_base="https://api.openai.com", + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert url == "https://api.openai.com/v1/evals/eval_123/cancel" + assert request_body == {} + + +def test_transform_cancel_eval_response(config: OpenAIEvalsConfig): + """Test transformation of cancel eval response""" + response = httpx.Response( + status_code=200, + json={ + "id": "eval_123", + "object": "eval", + "status": "cancelled", + }, + request=httpx.Request("POST", "https://api.openai.com/v1/evals/eval_123/cancel"), + ) + + result = config.transform_cancel_eval_response( + raw_response=response, + logging_obj=None, # type: ignore + ) + + assert result.id == "eval_123" + assert result.object == "eval"