From f77324d766958286cbaa1c6313f36c0ca98461e3 Mon Sep 17 00:00:00 2001 From: Mateo Wang <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 16 May 2026 03:21:28 +0000 Subject: [PATCH] refactor(tests): move reasoning_effort grid suite under llm_translation, drop v4 naming - Drop the "v4" suffix throughout: it referred to the QA sweep iteration, not this test suite. There's only one regression suite, so just call it reasoning_effort_grid. - Move tests/test_litellm/reasoning_effort_grid_v4/ -> tests/llm_translation/ reasoning_effort_grid/. Two reasons: 1. The parent tests/test_litellm/conftest.py installs an autouse fixture (isolate_host_aws_config) that clears every AWS_* env var before each test, which would silently skip every Bedrock cell. 2. tests/llm_translation/conftest.py already wires up the Redis-backed VCR persister and auto-applies @pytest.mark.vcr to every collected item via apply_vcr_auto_marker_to_items. Living under that conftest means the suite gets cassette replay for free -- first CI run with provider creds records 231 cassettes, every subsequent run replays them with no live spend. - Trim the suite's own conftest down to just the wire_capture fixture; the inherited llm_translation conftest covers the VCR plumbing. - Drop the dedicated reasoning_effort_grid_v4_e2e CircleCI job. The existing llm_translation_testing job globs tests/llm_translation/**/test_*.py, so the suite is gated by an existing job with no new wiring. --- .circleci/config.yml | 44 ------------------- .../reasoning_effort_grid}/__init__.py | 0 .../reasoning_effort_grid}/conftest.py | 30 +++++-------- .../reasoning_effort_grid}/grid_spec.py | 15 ++++--- .../test_reasoning_effort_grid.py} | 19 ++++---- 5 files changed, 31 insertions(+), 77 deletions(-) rename tests/{test_litellm/reasoning_effort_grid_v4 => llm_translation/reasoning_effort_grid}/__init__.py (100%) rename tests/{test_litellm/reasoning_effort_grid_v4 => llm_translation/reasoning_effort_grid}/conftest.py (61%) rename tests/{test_litellm/reasoning_effort_grid_v4 => llm_translation/reasoning_effort_grid}/grid_spec.py (94%) rename tests/{test_litellm/reasoning_effort_grid_v4/test_grid_v4.py => llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py} (91%) diff --git a/.circleci/config.yml b/.circleci/config.yml index 3dd1221c12f..9a5b6da77f3 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -576,48 +576,6 @@ jobs: no_output_timeout: 15m # Store test results - - store_test_results: - path: test-results - reasoning_effort_grid_v4_e2e: - docker: - - *python312_image - working_directory: ~/project - resource_class: large - - steps: - - checkout - - setup_google_dns - - install_uv - - restore_cache: - keys: - - v1-uv-cache-{{ checksum "uv.lock" }} - - run: - name: Install Dependencies - command: | - uv sync --frozen --all-groups --all-extras --python 3.12 - - save_cache: - paths: - - ~/.cache/uv - key: v1-uv-cache-{{ checksum "uv.lock" }} - # Grid v4 exercises reasoning_effort mapping against real Anthropic, - # Azure AI Foundry, Vertex AI, Bedrock Converse, and Bedrock Invoke - # endpoints. Per-route cells pytest-skip themselves when the matching - # provider env vars are absent, so PRs without credentials no-op. - - run: - name: Run reasoning_effort grid v4 e2e suite - command: | - mkdir -p test-results - uv run --no-sync python -m pytest \ - tests/test_litellm/reasoning_effort_grid_v4/ \ - -v \ - --junitxml=test-results/junit.xml \ - --durations=20 \ - -n 4 \ - --timeout=180 --timeout_method=thread \ - --retries 2 --retry-delay 5 \ - --max-worker-restart=5 - no_output_timeout: 20m - - store_test_results: path: test-results realtime_translation_testing: @@ -2661,8 +2619,6 @@ workflows: filters: *main_branches - llm_translation_testing: filters: *main_branches - - reasoning_effort_grid_v4_e2e: - filters: *main_branches - realtime_translation_testing: filters: *main_branches - agent_testing: diff --git a/tests/test_litellm/reasoning_effort_grid_v4/__init__.py b/tests/llm_translation/reasoning_effort_grid/__init__.py similarity index 100% rename from tests/test_litellm/reasoning_effort_grid_v4/__init__.py rename to tests/llm_translation/reasoning_effort_grid/__init__.py diff --git a/tests/test_litellm/reasoning_effort_grid_v4/conftest.py b/tests/llm_translation/reasoning_effort_grid/conftest.py similarity index 61% rename from tests/test_litellm/reasoning_effort_grid_v4/conftest.py rename to tests/llm_translation/reasoning_effort_grid/conftest.py index 5e1a4e0e974..aad85307cc6 100644 --- a/tests/test_litellm/reasoning_effort_grid_v4/conftest.py +++ b/tests/llm_translation/reasoning_effort_grid/conftest.py @@ -1,6 +1,12 @@ -"""Shared fixtures for the reasoning_effort grid v4 e2e suite.""" +"""Shared fixtures for the reasoning_effort grid e2e suite. + +VCR wiring (Redis-backed cassette persister, auto-application of +``@pytest.mark.vcr`` to every collected item, cassette-cache health summary) +is inherited from ``tests/llm_translation/conftest.py``. This file only +contributes the ``wire_capture`` fixture, which records the wire body +LiteLLM sends upstream so each cell can inspect it. +""" -import os from typing import Any, Dict, List, Optional import pytest @@ -12,10 +18,10 @@ from litellm.integrations.custom_logger import CustomLogger class _WireBodyCapture(CustomLogger): """Pre-call hook that records the outgoing wire body LiteLLM sends upstream. - `complete_input_dict` is the fully transformed provider request as set by - every provider transformation in `litellm/llms/**`. Capturing it here means - a regression anywhere in the transformation chain (strip, rename, drop) - surfaces as an assertion failure on the cell that depends on it. + ``complete_input_dict`` is the fully transformed provider request as set + by every provider transformation in ``litellm/llms/**``. Capturing it here + means a regression anywhere in the transformation chain (strip, rename, + drop) surfaces as an assertion failure on the cell that depends on it. """ def __init__(self) -> None: @@ -37,9 +43,6 @@ class _WireBodyCapture(CustomLogger): def latest(self) -> Optional[Dict[str, Any]]: return self.records[-1] if self.records else None - def reset(self) -> None: - self.records.clear() - @pytest.fixture() def wire_capture(): @@ -50,12 +53,3 @@ def wire_capture(): yield capture finally: litellm.callbacks = previous - - -@pytest.fixture(scope="session") -def vertex_credentials_path() -> Optional[str]: - """Resolve a usable Vertex credentials file path or None.""" - path = os.environ.get("GOOGLE_APPLICATION_CREDENTIALS") - if path and os.path.exists(path): - return path - return None diff --git a/tests/test_litellm/reasoning_effort_grid_v4/grid_spec.py b/tests/llm_translation/reasoning_effort_grid/grid_spec.py similarity index 94% rename from tests/test_litellm/reasoning_effort_grid_v4/grid_spec.py rename to tests/llm_translation/reasoning_effort_grid/grid_spec.py index 76243fe227b..5185dfc2daf 100644 --- a/tests/test_litellm/reasoning_effort_grid_v4/grid_spec.py +++ b/tests/llm_translation/reasoning_effort_grid/grid_spec.py @@ -1,12 +1,13 @@ """ -Canonical post-fix expectations for the reasoning_effort grid v4 sweep. +Canonical post-fix expectations for the reasoning_effort grid sweep. -The QA sweep on https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610 -covered 21 (provider x model) combos x 11 effort values (231 cells). The follow-up -PR https://github.com/BerriAI/litellm/pull/27074 closed nine bugs surfaced by that -sweep. This module encodes the post-fix expectations as a small rule set keyed by -(model_mode, effort) and per-model capability overrides, then expands them across -the model x effort matrix per route. +The original QA sweep on +https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610 +covered 21 (provider x model) combos x 11 effort values (231 cells). The +follow-up PR https://github.com/BerriAI/litellm/pull/27074 closed nine bugs +surfaced by that sweep. This module encodes the post-fix expectations as a +small rule set keyed by (model_mode, effort) and per-model capability +overrides, then expands them across the model x effort matrix per route. """ from dataclasses import dataclass, field diff --git a/tests/test_litellm/reasoning_effort_grid_v4/test_grid_v4.py b/tests/llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py similarity index 91% rename from tests/test_litellm/reasoning_effort_grid_v4/test_grid_v4.py rename to tests/llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py index 9eec572988b..6925a2aea92 100644 --- a/tests/test_litellm/reasoning_effort_grid_v4/test_grid_v4.py +++ b/tests/llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py @@ -1,6 +1,6 @@ """ -End-to-end grid v4 regression suite for reasoning_effort mapping across -Anthropic-backed routes. +End-to-end regression suite for reasoning_effort mapping across the +Anthropic-backed routes covered by the original QA sweep. Encodes the 21 (provider x model) x 11 effort matrix (231 cells) from the QA sweep on https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610 @@ -13,9 +13,12 @@ against. Each cell asserts: - Status code returned by LiteLLM (200 vs BadRequestError -> 400) -- the regression signal for clean-error vs leaked-500 mappings. -Hits real provider endpoints. Each route is skipped at runtime when its -required env vars are absent, so PR builds without provider credentials no-op -gracefully. +Calls go to real provider endpoints, but the parent +``tests/llm_translation/conftest.py`` auto-applies ``@pytest.mark.vcr`` to +every collected item, so first run records cassettes (Redis-backed) and +subsequent CI runs replay them with no live spend. Each route still skips at +runtime when its required env vars are absent, so PR builds without provider +credentials no-op gracefully. """ import os @@ -180,7 +183,7 @@ async def _call_messages( @pytest.mark.parametrize( ("route_name", "model", "effort", "cell"), _PARAMS, ids=_PARAM_IDS ) -async def test_reasoning_effort_grid_v4( +async def test_reasoning_effort_grid( route_name: str, model: ModelEntry, effort: str, @@ -209,7 +212,7 @@ async def test_reasoning_effort_grid_v4( raise -def test_grid_v4_cell_count() -> None: +def test_grid_cell_count() -> None: """Guard against accidental drops or duplicates in the grid spec.""" assert len(_PARAMS) == 21 * 11, ( f"expected 231 cells (21 provider x model combos x 11 efforts), " @@ -217,7 +220,7 @@ def test_grid_v4_cell_count() -> None: ) -def test_grid_v4_route_coverage() -> None: +def test_grid_route_coverage() -> None: """The grid must cover every route the original QA sweep covered.""" route_names = {route.name for route in ROUTES} assert route_names == {