mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
test(harness): scaffold tests/harness_suites for the 4h harness migration
Per the keep/drop audit (8c): suite directories with a manifest.yaml stub mirroring the tests/claude_code compat-matrix conventions (suite id matches directory name; compat_result tagged-union fixture contract). The conftest puts tests/llm_translation and tests/llm_responses_api_testing on sys.path so moved tests keep importing their shared bases (base_llm_unit_tests, base_responses_api) from their original homes. Non-chat live tests (embeddings, rerank, transcription, TTS, agents, skills, evals) are NOT moved; HANDOFF.md lists them for Sameer's suites per the scope split.
This commit is contained in:
parent
417b57e8ef
commit
97a34857f9
17 changed files with 142 additions and 111 deletions
45
tests/harness_suites/HANDOFF.md
Normal file
45
tests/harness_suites/HANDOFF.md
Normal file
|
|
@ -0,0 +1,45 @@
|
|||
# Non-chat live tests: handoff to Sameer
|
||||
|
||||
The chat-scope keep/drop audit (reports/ci-auditor.md, 8c) identified these
|
||||
live tests as harness material, but they are non-chat surfaces (embeddings,
|
||||
rerank, transcription, TTS, agents, skills, evals) and therefore Sameer's
|
||||
scope per 02-mateo-scope.md "Out of scope". They were NOT moved into
|
||||
tests/harness_suites; they stay where they are until Sameer builds his
|
||||
suites. Nothing here blocks the chat suites.
|
||||
|
||||
## Whole files (still in tests/llm_translation/)
|
||||
|
||||
| file | what it is |
|
||||
|---|---|
|
||||
| test_containers_api.py | live OpenAI Containers API CRUD |
|
||||
| test_evals_api.py | live OpenAI Evals CRUD via VCR cassettes |
|
||||
| test_deepgram.py | live audio-transcription base subclass |
|
||||
| test_hosted_vllm_embedding_e2e.py | live vLLM embeddings, env-gated |
|
||||
| test_jina_ai.py | live rerank base subclass + live embedding |
|
||||
| test_skills_api.py | live Anthropic Skills CRUD base class. NOTE: no concrete subclass exists, so it executes nothing in CI today (audit 9); decide whether to subclass or delete |
|
||||
| test_skills_e2e.py | skipped local-only e2e: real DB + live model through skills hook |
|
||||
|
||||
## Live portions of files that otherwise stayed as unit tests
|
||||
|
||||
| file | live tests for Sameer |
|
||||
|---|---|
|
||||
| test_azure_agents.py | test_azure_ai_agents_acompletion_non_streaming / _streaming / _conversation_continuity |
|
||||
| test_bedrock_agentcore.py | test_bedrock_agentcore_basic, test_bedrock_agentcore_with_streaming |
|
||||
| test_bedrock_agents.py | test_bedrock_agents, test_bedrock_agents_with_streaming (both @skip live) |
|
||||
| test_bedrock_completion.py | TestBedrockRerank, TestBedrockCohereRerank, TestBedrockEmbedding |
|
||||
| test_bedrock_embedding.py | test_e2e_bedrock_embedding, test_e2e_bedrock_embedding_image_twelvelabs_marengo |
|
||||
| test_bedrock_nova_embedding.py | TestNovaEmbeddingIntegration (all @skip) |
|
||||
| test_cohere.py | test_cohere_embedding_outout_dimensions + the 8 test_cohere_embed_v4_* live tests |
|
||||
| test_azure_openai.py | TestAzureEmbedding |
|
||||
| test_openai.py | TestOpenAIGPT4OAudioTranscription |
|
||||
| test_elevenlabs.py | TestElevenLabsAudioTranscription inherited live transcription tests |
|
||||
| test_fireworks_ai_translation.py | TestFireworksAIAudioTranscription |
|
||||
| test_minimax_tts.py | test_speech_basic, test_speech_with_custom_params (placeholder-key live) |
|
||||
| test_rerank.py | test_basic_rerank, test_rerank_custom_callbacks, test_basic_rerank_caching, test_rerank_cohere_api, test_basic_rerank_together_ai |
|
||||
| test_voyage_ai.py | TestVoyageAI base subclass |
|
||||
| test_gemini.py | test_gemini_embedding |
|
||||
| test_text_completion.py | test_text_completion_include_usage |
|
||||
|
||||
These still run in `llm_translation_testing` per-commit until they migrate;
|
||||
they are the residual live (network) load in that job. The remaining
|
||||
@pytest.mark.flaky markers in the tree sit on these tests only.
|
||||
64
tests/harness_suites/conftest.py
Normal file
64
tests/harness_suites/conftest.py
Normal file
|
|
@ -0,0 +1,64 @@
|
|||
"""Pytest plumbing for the 4h-harness suites.
|
||||
|
||||
These suites hold the live provider tests migrated out of the per-commit
|
||||
CircleCI jobs (`llm_translation_testing`, `llm_responses_api_testing`) by the
|
||||
chat-scope keep/drop audit. The harness owns scheduling, deploy, credentials,
|
||||
rate limiting, result merging, and alerting; suite authors only write tests.
|
||||
|
||||
The moved tests still import shared bases and helpers from their original
|
||||
directories (`base_llm_unit_tests`, `base_responses_api`, sibling test
|
||||
modules), so both directories are put on sys.path here.
|
||||
|
||||
`compat_result` mirrors the tagged-union recorder contract from
|
||||
tests/claude_code/conftest.py (the compat matrix, suite #1 on the harness).
|
||||
This is a stub: it validates and stores results but the merge-to-
|
||||
`compat-results.json` hook belongs to the harness-engineer's contract and is
|
||||
not duplicated here. Tests that do not call it are reported by pass/fail as
|
||||
usual.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import pytest
|
||||
|
||||
_REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", ".."))
|
||||
for _p in (
|
||||
_REPO_ROOT,
|
||||
os.path.join(_REPO_ROOT, "tests", "llm_translation"),
|
||||
os.path.join(_REPO_ROOT, "tests", "llm_responses_api_testing"),
|
||||
):
|
||||
if _p not in sys.path:
|
||||
sys.path.insert(0, _p)
|
||||
|
||||
VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
|
||||
|
||||
|
||||
@dataclass
|
||||
class CompatResult:
|
||||
value: Optional[Dict[str, Any]] = None
|
||||
values: List[Dict[str, Any]] = field(default_factory=list)
|
||||
|
||||
def set(self, result: Dict[str, Any]) -> None:
|
||||
self.value = self._validate(result)
|
||||
|
||||
def add(self, result: Dict[str, Any]) -> None:
|
||||
self.values.append(self._validate(result))
|
||||
|
||||
@staticmethod
|
||||
def _validate(result: Dict[str, Any]) -> Dict[str, Any]:
|
||||
status = result.get("status") if isinstance(result, dict) else None
|
||||
if status not in VALID_STATUSES:
|
||||
raise ValueError(
|
||||
f"compat_result status must be one of {sorted(VALID_STATUSES)}, got {status!r}"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def compat_result() -> CompatResult:
|
||||
return CompatResult()
|
||||
33
tests/harness_suites/manifest.yaml
Normal file
33
tests/harness_suites/manifest.yaml
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
# 4h-harness suite manifest -- stub.
|
||||
#
|
||||
# Mirrors the conventions of tests/claude_code/manifest.yaml (the compat
|
||||
# matrix, suite #1 on the harness): `id` MUST match the suite directory name
|
||||
# on disk; the harness infers the suite for each test from its file path.
|
||||
# The harness-engineer owns the schema; entries here register the suites
|
||||
# created by the chat-scope CircleCI keep/drop migration (audit 8c).
|
||||
#
|
||||
# Every test in these suites is live provider truth: env API keys, no
|
||||
# patching of the asserted layer. The harness owns retry policy and
|
||||
# per-provider rate limiting; the @pytest.mark.flaky markers these tests
|
||||
# carried in CircleCI were stripped at migration time.
|
||||
|
||||
schema_version: "1"
|
||||
|
||||
suites:
|
||||
- id: chat_live_anthropic
|
||||
name: Chat live (Anthropic)
|
||||
# Largely overlaps the compat matrix feature rows (tools, json mode,
|
||||
# pdf, images, caching, streaming); dedupe against
|
||||
# tests/claude_code/manifest.yaml before adding cells.
|
||||
- id: chat_live_bedrock
|
||||
name: Chat live (Bedrock converse + invoke)
|
||||
- id: chat_live_azure
|
||||
name: Chat live (Azure OpenAI + Azure AI)
|
||||
- id: chat_live_openai
|
||||
name: Chat live (OpenAI)
|
||||
- id: chat_live_gemini
|
||||
name: Chat live (Google AI Studio)
|
||||
- id: chat_live_longtail
|
||||
name: Chat live (long-tail providers, env-gated cells)
|
||||
- id: responses_api_live
|
||||
name: Responses API live (OpenAI, Anthropic, Azure, Gemini)
|
||||
|
|
@ -1,111 +0,0 @@
|
|||
import os
|
||||
import sys
|
||||
import pytest
|
||||
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../.."))
|
||||
import litellm
|
||||
from base_responses_api import BaseResponsesAPITest
|
||||
from openai.types.responses.function_tool import FunctionTool
|
||||
|
||||
|
||||
class TestAnthropicResponsesAPITest(BaseResponsesAPITest):
|
||||
def get_base_completion_call_args(self):
|
||||
# litellm._turn_on_debug()
|
||||
return {
|
||||
"model": "anthropic/claude-sonnet-4-5",
|
||||
}
|
||||
|
||||
async def test_basic_openai_responses_delete_endpoint(self, sync_mode=False):
|
||||
pytest.skip("DELETE responses is not supported for anthropic")
|
||||
|
||||
async def test_basic_openai_responses_streaming_delete_endpoint(
|
||||
self, sync_mode=False
|
||||
):
|
||||
pytest.skip("DELETE responses is not supported for anthropic")
|
||||
|
||||
async def test_basic_openai_responses_get_endpoint(self, sync_mode=False):
|
||||
pytest.skip("GET responses is not supported for anthropic")
|
||||
|
||||
async def test_basic_openai_responses_cancel_endpoint(self, sync_mode=False):
|
||||
pytest.skip("CANCEL responses is not supported for anthropic")
|
||||
|
||||
async def test_cancel_responses_invalid_response_id(self, sync_mode=False):
|
||||
pytest.skip("CANCEL responses is not supported for anthropic")
|
||||
|
||||
|
||||
def test_multiturn_tool_calls():
|
||||
# Test streaming response with tools for Anthropic
|
||||
litellm._turn_on_debug()
|
||||
shell_tool = dict(
|
||||
FunctionTool(
|
||||
type="function",
|
||||
name="shell",
|
||||
description="Runs a shell command, and returns its output.",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"command": {"type": "array", "items": {"type": "string"}},
|
||||
"workdir": {
|
||||
"type": "string",
|
||||
"description": "The working directory for the command.",
|
||||
},
|
||||
},
|
||||
"required": ["command"],
|
||||
},
|
||||
strict=True,
|
||||
)
|
||||
)
|
||||
|
||||
# Step 1: Initial request with the tool
|
||||
response = litellm.responses(
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "input_text", "text": "make a hello world html file"}
|
||||
],
|
||||
"type": "message",
|
||||
}
|
||||
],
|
||||
model="anthropic/claude-haiku-4-5-20251001",
|
||||
instructions="You are a helpful coding assistant.",
|
||||
tools=[shell_tool],
|
||||
)
|
||||
|
||||
print("response=", response)
|
||||
|
||||
# Step 2: Send the results of the tool call back to the model
|
||||
# Get the response ID and tool call ID from the response
|
||||
|
||||
response_id = response.id
|
||||
tool_call_id = None
|
||||
for item in response.output:
|
||||
if hasattr(item, "type") and item.type == "function_call":
|
||||
tool_call_id = getattr(item, "call_id", None)
|
||||
if tool_call_id:
|
||||
break
|
||||
|
||||
# Validate that we got a tool call with a valid call_id
|
||||
if not tool_call_id:
|
||||
raise AssertionError(
|
||||
f"Expected a function_call with a valid call_id in response.output, but got: {response.output}"
|
||||
)
|
||||
|
||||
# Use await with asyncio.run for the async function
|
||||
follow_up_response = litellm.responses(
|
||||
model="anthropic/claude-haiku-4-5-20251001",
|
||||
previous_response_id=response_id,
|
||||
input=[
|
||||
{
|
||||
"type": "function_call_output",
|
||||
"call_id": tool_call_id,
|
||||
"output": '{"output":"<html>\\n<head>\\n <title>Hello Page</title>\\n</head>\\n<body>\\n <h1>Hi</h1>\\n <p>Welcome to this simple webpage!</p>\\n</body>\\n</html> > index.html\\n","metadata":{"exit_code":0,"duration_seconds":0}}',
|
||||
}
|
||||
],
|
||||
tools=[shell_tool],
|
||||
)
|
||||
|
||||
print("follow_up_response=", follow_up_response)
|
||||
|
||||
|
||||
Loading…
Add table
Reference in a new issue