From 97a34857f9015c8f448b1eac7c9d5597afd2a577 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 11 Jun 2026 19:14:15 +0000 Subject: [PATCH] test(harness): scaffold tests/harness_suites for the 4h harness migration Per the keep/drop audit (8c): suite directories with a manifest.yaml stub mirroring the tests/claude_code compat-matrix conventions (suite id matches directory name; compat_result tagged-union fixture contract). The conftest puts tests/llm_translation and tests/llm_responses_api_testing on sys.path so moved tests keep importing their shared bases (base_llm_unit_tests, base_responses_api) from their original homes. Non-chat live tests (embeddings, rerank, transcription, TTS, agents, skills, evals) are NOT moved; HANDOFF.md lists them for Sameer's suites per the scope split. --- tests/harness_suites/HANDOFF.md | 45 +++++++ .../reasoning_effort_grid/__init__.py | 0 .../reasoning_effort_grid/conftest.py | 0 .../reasoning_effort_grid/grid_spec.py | 0 .../test_reasoning_effort_grid.py | 0 .../chat_live_bedrock}/test_bedrock_llama.py | 0 .../test_bedrock_nova_json.py | 0 .../chat_live_longtail}/test_a2a.py | 0 .../chat_live_longtail}/test_mistral_api.py | 0 .../chat_live_longtail}/test_openrouter.py | 0 .../chat_live_longtail}/test_snowflake.py | 0 .../chat_live_openai}/test_gpt4o_audio.py | 0 .../chat_live_openai}/test_prompt_caching.py | 0 .../test_router_llm_translation_tests.py | 0 tests/harness_suites/conftest.py | 64 ++++++++++ tests/harness_suites/manifest.yaml | 33 ++++++ .../test_anthropic_responses_api.py | 111 ------------------ 17 files changed, 142 insertions(+), 111 deletions(-) create mode 100644 tests/harness_suites/HANDOFF.md rename tests/{llm_translation => harness_suites/chat_live_bedrock}/reasoning_effort_grid/__init__.py (100%) rename tests/{llm_translation => harness_suites/chat_live_bedrock}/reasoning_effort_grid/conftest.py (100%) rename tests/{llm_translation => harness_suites/chat_live_bedrock}/reasoning_effort_grid/grid_spec.py (100%) rename tests/{llm_translation => harness_suites/chat_live_bedrock}/reasoning_effort_grid/test_reasoning_effort_grid.py (100%) rename tests/{llm_translation => harness_suites/chat_live_bedrock}/test_bedrock_llama.py (100%) rename tests/{llm_translation => harness_suites/chat_live_bedrock}/test_bedrock_nova_json.py (100%) rename tests/{llm_translation => harness_suites/chat_live_longtail}/test_a2a.py (100%) rename tests/{llm_translation => harness_suites/chat_live_longtail}/test_mistral_api.py (100%) rename tests/{llm_translation => harness_suites/chat_live_longtail}/test_openrouter.py (100%) rename tests/{llm_translation => harness_suites/chat_live_longtail}/test_snowflake.py (100%) rename tests/{llm_translation => harness_suites/chat_live_openai}/test_gpt4o_audio.py (100%) rename tests/{llm_translation => harness_suites/chat_live_openai}/test_prompt_caching.py (100%) rename tests/{llm_translation => harness_suites/chat_live_openai}/test_router_llm_translation_tests.py (100%) create mode 100644 tests/harness_suites/conftest.py create mode 100644 tests/harness_suites/manifest.yaml delete mode 100644 tests/llm_responses_api_testing/test_anthropic_responses_api.py diff --git a/tests/harness_suites/HANDOFF.md b/tests/harness_suites/HANDOFF.md new file mode 100644 index 00000000000..90ffb2298af --- /dev/null +++ b/tests/harness_suites/HANDOFF.md @@ -0,0 +1,45 @@ +# Non-chat live tests: handoff to Sameer + +The chat-scope keep/drop audit (reports/ci-auditor.md, 8c) identified these +live tests as harness material, but they are non-chat surfaces (embeddings, +rerank, transcription, TTS, agents, skills, evals) and therefore Sameer's +scope per 02-mateo-scope.md "Out of scope". They were NOT moved into +tests/harness_suites; they stay where they are until Sameer builds his +suites. Nothing here blocks the chat suites. + +## Whole files (still in tests/llm_translation/) + +| file | what it is | +|---|---| +| test_containers_api.py | live OpenAI Containers API CRUD | +| test_evals_api.py | live OpenAI Evals CRUD via VCR cassettes | +| test_deepgram.py | live audio-transcription base subclass | +| test_hosted_vllm_embedding_e2e.py | live vLLM embeddings, env-gated | +| test_jina_ai.py | live rerank base subclass + live embedding | +| test_skills_api.py | live Anthropic Skills CRUD base class. NOTE: no concrete subclass exists, so it executes nothing in CI today (audit 9); decide whether to subclass or delete | +| test_skills_e2e.py | skipped local-only e2e: real DB + live model through skills hook | + +## Live portions of files that otherwise stayed as unit tests + +| file | live tests for Sameer | +|---|---| +| test_azure_agents.py | test_azure_ai_agents_acompletion_non_streaming / _streaming / _conversation_continuity | +| test_bedrock_agentcore.py | test_bedrock_agentcore_basic, test_bedrock_agentcore_with_streaming | +| test_bedrock_agents.py | test_bedrock_agents, test_bedrock_agents_with_streaming (both @skip live) | +| test_bedrock_completion.py | TestBedrockRerank, TestBedrockCohereRerank, TestBedrockEmbedding | +| test_bedrock_embedding.py | test_e2e_bedrock_embedding, test_e2e_bedrock_embedding_image_twelvelabs_marengo | +| test_bedrock_nova_embedding.py | TestNovaEmbeddingIntegration (all @skip) | +| test_cohere.py | test_cohere_embedding_outout_dimensions + the 8 test_cohere_embed_v4_* live tests | +| test_azure_openai.py | TestAzureEmbedding | +| test_openai.py | TestOpenAIGPT4OAudioTranscription | +| test_elevenlabs.py | TestElevenLabsAudioTranscription inherited live transcription tests | +| test_fireworks_ai_translation.py | TestFireworksAIAudioTranscription | +| test_minimax_tts.py | test_speech_basic, test_speech_with_custom_params (placeholder-key live) | +| test_rerank.py | test_basic_rerank, test_rerank_custom_callbacks, test_basic_rerank_caching, test_rerank_cohere_api, test_basic_rerank_together_ai | +| test_voyage_ai.py | TestVoyageAI base subclass | +| test_gemini.py | test_gemini_embedding | +| test_text_completion.py | test_text_completion_include_usage | + +These still run in `llm_translation_testing` per-commit until they migrate; +they are the residual live (network) load in that job. The remaining +@pytest.mark.flaky markers in the tree sit on these tests only. diff --git a/tests/llm_translation/reasoning_effort_grid/__init__.py b/tests/harness_suites/chat_live_bedrock/reasoning_effort_grid/__init__.py similarity index 100% rename from tests/llm_translation/reasoning_effort_grid/__init__.py rename to tests/harness_suites/chat_live_bedrock/reasoning_effort_grid/__init__.py diff --git a/tests/llm_translation/reasoning_effort_grid/conftest.py b/tests/harness_suites/chat_live_bedrock/reasoning_effort_grid/conftest.py similarity index 100% rename from tests/llm_translation/reasoning_effort_grid/conftest.py rename to tests/harness_suites/chat_live_bedrock/reasoning_effort_grid/conftest.py diff --git a/tests/llm_translation/reasoning_effort_grid/grid_spec.py b/tests/harness_suites/chat_live_bedrock/reasoning_effort_grid/grid_spec.py similarity index 100% rename from tests/llm_translation/reasoning_effort_grid/grid_spec.py rename to tests/harness_suites/chat_live_bedrock/reasoning_effort_grid/grid_spec.py diff --git a/tests/llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py b/tests/harness_suites/chat_live_bedrock/reasoning_effort_grid/test_reasoning_effort_grid.py similarity index 100% rename from tests/llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py rename to tests/harness_suites/chat_live_bedrock/reasoning_effort_grid/test_reasoning_effort_grid.py diff --git a/tests/llm_translation/test_bedrock_llama.py b/tests/harness_suites/chat_live_bedrock/test_bedrock_llama.py similarity index 100% rename from tests/llm_translation/test_bedrock_llama.py rename to tests/harness_suites/chat_live_bedrock/test_bedrock_llama.py diff --git a/tests/llm_translation/test_bedrock_nova_json.py b/tests/harness_suites/chat_live_bedrock/test_bedrock_nova_json.py similarity index 100% rename from tests/llm_translation/test_bedrock_nova_json.py rename to tests/harness_suites/chat_live_bedrock/test_bedrock_nova_json.py diff --git a/tests/llm_translation/test_a2a.py b/tests/harness_suites/chat_live_longtail/test_a2a.py similarity index 100% rename from tests/llm_translation/test_a2a.py rename to tests/harness_suites/chat_live_longtail/test_a2a.py diff --git a/tests/llm_translation/test_mistral_api.py b/tests/harness_suites/chat_live_longtail/test_mistral_api.py similarity index 100% rename from tests/llm_translation/test_mistral_api.py rename to tests/harness_suites/chat_live_longtail/test_mistral_api.py diff --git a/tests/llm_translation/test_openrouter.py b/tests/harness_suites/chat_live_longtail/test_openrouter.py similarity index 100% rename from tests/llm_translation/test_openrouter.py rename to tests/harness_suites/chat_live_longtail/test_openrouter.py diff --git a/tests/llm_translation/test_snowflake.py b/tests/harness_suites/chat_live_longtail/test_snowflake.py similarity index 100% rename from tests/llm_translation/test_snowflake.py rename to tests/harness_suites/chat_live_longtail/test_snowflake.py diff --git a/tests/llm_translation/test_gpt4o_audio.py b/tests/harness_suites/chat_live_openai/test_gpt4o_audio.py similarity index 100% rename from tests/llm_translation/test_gpt4o_audio.py rename to tests/harness_suites/chat_live_openai/test_gpt4o_audio.py diff --git a/tests/llm_translation/test_prompt_caching.py b/tests/harness_suites/chat_live_openai/test_prompt_caching.py similarity index 100% rename from tests/llm_translation/test_prompt_caching.py rename to tests/harness_suites/chat_live_openai/test_prompt_caching.py diff --git a/tests/llm_translation/test_router_llm_translation_tests.py b/tests/harness_suites/chat_live_openai/test_router_llm_translation_tests.py similarity index 100% rename from tests/llm_translation/test_router_llm_translation_tests.py rename to tests/harness_suites/chat_live_openai/test_router_llm_translation_tests.py diff --git a/tests/harness_suites/conftest.py b/tests/harness_suites/conftest.py new file mode 100644 index 00000000000..3cd7309edd4 --- /dev/null +++ b/tests/harness_suites/conftest.py @@ -0,0 +1,64 @@ +"""Pytest plumbing for the 4h-harness suites. + +These suites hold the live provider tests migrated out of the per-commit +CircleCI jobs (`llm_translation_testing`, `llm_responses_api_testing`) by the +chat-scope keep/drop audit. The harness owns scheduling, deploy, credentials, +rate limiting, result merging, and alerting; suite authors only write tests. + +The moved tests still import shared bases and helpers from their original +directories (`base_llm_unit_tests`, `base_responses_api`, sibling test +modules), so both directories are put on sys.path here. + +`compat_result` mirrors the tagged-union recorder contract from +tests/claude_code/conftest.py (the compat matrix, suite #1 on the harness). +This is a stub: it validates and stores results but the merge-to- +`compat-results.json` hook belongs to the harness-engineer's contract and is +not duplicated here. Tests that do not call it are reported by pass/fail as +usual. +""" + +from __future__ import annotations + +import os +import sys +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional + +import pytest + +_REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")) +for _p in ( + _REPO_ROOT, + os.path.join(_REPO_ROOT, "tests", "llm_translation"), + os.path.join(_REPO_ROOT, "tests", "llm_responses_api_testing"), +): + if _p not in sys.path: + sys.path.insert(0, _p) + +VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"} + + +@dataclass +class CompatResult: + value: Optional[Dict[str, Any]] = None + values: List[Dict[str, Any]] = field(default_factory=list) + + def set(self, result: Dict[str, Any]) -> None: + self.value = self._validate(result) + + def add(self, result: Dict[str, Any]) -> None: + self.values.append(self._validate(result)) + + @staticmethod + def _validate(result: Dict[str, Any]) -> Dict[str, Any]: + status = result.get("status") if isinstance(result, dict) else None + if status not in VALID_STATUSES: + raise ValueError( + f"compat_result status must be one of {sorted(VALID_STATUSES)}, got {status!r}" + ) + return result + + +@pytest.fixture +def compat_result() -> CompatResult: + return CompatResult() diff --git a/tests/harness_suites/manifest.yaml b/tests/harness_suites/manifest.yaml new file mode 100644 index 00000000000..573301fbe3c --- /dev/null +++ b/tests/harness_suites/manifest.yaml @@ -0,0 +1,33 @@ +# 4h-harness suite manifest -- stub. +# +# Mirrors the conventions of tests/claude_code/manifest.yaml (the compat +# matrix, suite #1 on the harness): `id` MUST match the suite directory name +# on disk; the harness infers the suite for each test from its file path. +# The harness-engineer owns the schema; entries here register the suites +# created by the chat-scope CircleCI keep/drop migration (audit 8c). +# +# Every test in these suites is live provider truth: env API keys, no +# patching of the asserted layer. The harness owns retry policy and +# per-provider rate limiting; the @pytest.mark.flaky markers these tests +# carried in CircleCI were stripped at migration time. + +schema_version: "1" + +suites: + - id: chat_live_anthropic + name: Chat live (Anthropic) + # Largely overlaps the compat matrix feature rows (tools, json mode, + # pdf, images, caching, streaming); dedupe against + # tests/claude_code/manifest.yaml before adding cells. + - id: chat_live_bedrock + name: Chat live (Bedrock converse + invoke) + - id: chat_live_azure + name: Chat live (Azure OpenAI + Azure AI) + - id: chat_live_openai + name: Chat live (OpenAI) + - id: chat_live_gemini + name: Chat live (Google AI Studio) + - id: chat_live_longtail + name: Chat live (long-tail providers, env-gated cells) + - id: responses_api_live + name: Responses API live (OpenAI, Anthropic, Azure, Gemini) diff --git a/tests/llm_responses_api_testing/test_anthropic_responses_api.py b/tests/llm_responses_api_testing/test_anthropic_responses_api.py deleted file mode 100644 index 4b04e07c178..00000000000 --- a/tests/llm_responses_api_testing/test_anthropic_responses_api.py +++ /dev/null @@ -1,111 +0,0 @@ -import os -import sys -import pytest - - -sys.path.insert(0, os.path.abspath("../..")) -import litellm -from base_responses_api import BaseResponsesAPITest -from openai.types.responses.function_tool import FunctionTool - - -class TestAnthropicResponsesAPITest(BaseResponsesAPITest): - def get_base_completion_call_args(self): - # litellm._turn_on_debug() - return { - "model": "anthropic/claude-sonnet-4-5", - } - - async def test_basic_openai_responses_delete_endpoint(self, sync_mode=False): - pytest.skip("DELETE responses is not supported for anthropic") - - async def test_basic_openai_responses_streaming_delete_endpoint( - self, sync_mode=False - ): - pytest.skip("DELETE responses is not supported for anthropic") - - async def test_basic_openai_responses_get_endpoint(self, sync_mode=False): - pytest.skip("GET responses is not supported for anthropic") - - async def test_basic_openai_responses_cancel_endpoint(self, sync_mode=False): - pytest.skip("CANCEL responses is not supported for anthropic") - - async def test_cancel_responses_invalid_response_id(self, sync_mode=False): - pytest.skip("CANCEL responses is not supported for anthropic") - - -def test_multiturn_tool_calls(): - # Test streaming response with tools for Anthropic - litellm._turn_on_debug() - shell_tool = dict( - FunctionTool( - type="function", - name="shell", - description="Runs a shell command, and returns its output.", - parameters={ - "type": "object", - "properties": { - "command": {"type": "array", "items": {"type": "string"}}, - "workdir": { - "type": "string", - "description": "The working directory for the command.", - }, - }, - "required": ["command"], - }, - strict=True, - ) - ) - - # Step 1: Initial request with the tool - response = litellm.responses( - input=[ - { - "role": "user", - "content": [ - {"type": "input_text", "text": "make a hello world html file"} - ], - "type": "message", - } - ], - model="anthropic/claude-haiku-4-5-20251001", - instructions="You are a helpful coding assistant.", - tools=[shell_tool], - ) - - print("response=", response) - - # Step 2: Send the results of the tool call back to the model - # Get the response ID and tool call ID from the response - - response_id = response.id - tool_call_id = None - for item in response.output: - if hasattr(item, "type") and item.type == "function_call": - tool_call_id = getattr(item, "call_id", None) - if tool_call_id: - break - - # Validate that we got a tool call with a valid call_id - if not tool_call_id: - raise AssertionError( - f"Expected a function_call with a valid call_id in response.output, but got: {response.output}" - ) - - # Use await with asyncio.run for the async function - follow_up_response = litellm.responses( - model="anthropic/claude-haiku-4-5-20251001", - previous_response_id=response_id, - input=[ - { - "type": "function_call_output", - "call_id": tool_call_id, - "output": '{"output":"\\n
\\nWelcome to this simple webpage!
\\n\\n > index.html\\n","metadata":{"exit_code":0,"duration_seconds":0}}', - } - ], - tools=[shell_tool], - ) - - print("follow_up_response=", follow_up_response) - -