diff --git a/.circleci/config.yml b/.circleci/config.yml index 3dd1221c12f..9a5b6da77f3 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -576,48 +576,6 @@ jobs: no_output_timeout: 15m # Store test results - - store_test_results: - path: test-results - reasoning_effort_grid_v4_e2e: - docker: - - *python312_image - working_directory: ~/project - resource_class: large - - steps: - - checkout - - setup_google_dns - - install_uv - - restore_cache: - keys: - - v1-uv-cache-{{ checksum "uv.lock" }} - - run: - name: Install Dependencies - command: | - uv sync --frozen --all-groups --all-extras --python 3.12 - - save_cache: - paths: - - ~/.cache/uv - key: v1-uv-cache-{{ checksum "uv.lock" }} - # Grid v4 exercises reasoning_effort mapping against real Anthropic, - # Azure AI Foundry, Vertex AI, Bedrock Converse, and Bedrock Invoke - # endpoints. Per-route cells pytest-skip themselves when the matching - # provider env vars are absent, so PRs without credentials no-op. - - run: - name: Run reasoning_effort grid v4 e2e suite - command: | - mkdir -p test-results - uv run --no-sync python -m pytest \ - tests/test_litellm/reasoning_effort_grid_v4/ \ - -v \ - --junitxml=test-results/junit.xml \ - --durations=20 \ - -n 4 \ - --timeout=180 --timeout_method=thread \ - --retries 2 --retry-delay 5 \ - --max-worker-restart=5 - no_output_timeout: 20m - - store_test_results: path: test-results realtime_translation_testing: @@ -2661,8 +2619,6 @@ workflows: filters: *main_branches - llm_translation_testing: filters: *main_branches - - reasoning_effort_grid_v4_e2e: - filters: *main_branches - realtime_translation_testing: filters: *main_branches - agent_testing: diff --git a/tests/test_litellm/reasoning_effort_grid_v4/__init__.py b/tests/llm_translation/reasoning_effort_grid/__init__.py similarity index 100% rename from tests/test_litellm/reasoning_effort_grid_v4/__init__.py rename to tests/llm_translation/reasoning_effort_grid/__init__.py diff --git a/tests/test_litellm/reasoning_effort_grid_v4/conftest.py b/tests/llm_translation/reasoning_effort_grid/conftest.py similarity index 61% rename from tests/test_litellm/reasoning_effort_grid_v4/conftest.py rename to tests/llm_translation/reasoning_effort_grid/conftest.py index 5e1a4e0e974..aad85307cc6 100644 --- a/tests/test_litellm/reasoning_effort_grid_v4/conftest.py +++ b/tests/llm_translation/reasoning_effort_grid/conftest.py @@ -1,6 +1,12 @@ -"""Shared fixtures for the reasoning_effort grid v4 e2e suite.""" +"""Shared fixtures for the reasoning_effort grid e2e suite. + +VCR wiring (Redis-backed cassette persister, auto-application of +``@pytest.mark.vcr`` to every collected item, cassette-cache health summary) +is inherited from ``tests/llm_translation/conftest.py``. This file only +contributes the ``wire_capture`` fixture, which records the wire body +LiteLLM sends upstream so each cell can inspect it. +""" -import os from typing import Any, Dict, List, Optional import pytest @@ -12,10 +18,10 @@ from litellm.integrations.custom_logger import CustomLogger class _WireBodyCapture(CustomLogger): """Pre-call hook that records the outgoing wire body LiteLLM sends upstream. - `complete_input_dict` is the fully transformed provider request as set by - every provider transformation in `litellm/llms/**`. Capturing it here means - a regression anywhere in the transformation chain (strip, rename, drop) - surfaces as an assertion failure on the cell that depends on it. + ``complete_input_dict`` is the fully transformed provider request as set + by every provider transformation in ``litellm/llms/**``. Capturing it here + means a regression anywhere in the transformation chain (strip, rename, + drop) surfaces as an assertion failure on the cell that depends on it. """ def __init__(self) -> None: @@ -37,9 +43,6 @@ class _WireBodyCapture(CustomLogger): def latest(self) -> Optional[Dict[str, Any]]: return self.records[-1] if self.records else None - def reset(self) -> None: - self.records.clear() - @pytest.fixture() def wire_capture(): @@ -50,12 +53,3 @@ def wire_capture(): yield capture finally: litellm.callbacks = previous - - -@pytest.fixture(scope="session") -def vertex_credentials_path() -> Optional[str]: - """Resolve a usable Vertex credentials file path or None.""" - path = os.environ.get("GOOGLE_APPLICATION_CREDENTIALS") - if path and os.path.exists(path): - return path - return None diff --git a/tests/test_litellm/reasoning_effort_grid_v4/grid_spec.py b/tests/llm_translation/reasoning_effort_grid/grid_spec.py similarity index 94% rename from tests/test_litellm/reasoning_effort_grid_v4/grid_spec.py rename to tests/llm_translation/reasoning_effort_grid/grid_spec.py index 76243fe227b..5185dfc2daf 100644 --- a/tests/test_litellm/reasoning_effort_grid_v4/grid_spec.py +++ b/tests/llm_translation/reasoning_effort_grid/grid_spec.py @@ -1,12 +1,13 @@ """ -Canonical post-fix expectations for the reasoning_effort grid v4 sweep. +Canonical post-fix expectations for the reasoning_effort grid sweep. -The QA sweep on https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610 -covered 21 (provider x model) combos x 11 effort values (231 cells). The follow-up -PR https://github.com/BerriAI/litellm/pull/27074 closed nine bugs surfaced by that -sweep. This module encodes the post-fix expectations as a small rule set keyed by -(model_mode, effort) and per-model capability overrides, then expands them across -the model x effort matrix per route. +The original QA sweep on +https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610 +covered 21 (provider x model) combos x 11 effort values (231 cells). The +follow-up PR https://github.com/BerriAI/litellm/pull/27074 closed nine bugs +surfaced by that sweep. This module encodes the post-fix expectations as a +small rule set keyed by (model_mode, effort) and per-model capability +overrides, then expands them across the model x effort matrix per route. """ from dataclasses import dataclass, field diff --git a/tests/test_litellm/reasoning_effort_grid_v4/test_grid_v4.py b/tests/llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py similarity index 91% rename from tests/test_litellm/reasoning_effort_grid_v4/test_grid_v4.py rename to tests/llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py index 9eec572988b..6925a2aea92 100644 --- a/tests/test_litellm/reasoning_effort_grid_v4/test_grid_v4.py +++ b/tests/llm_translation/reasoning_effort_grid/test_reasoning_effort_grid.py @@ -1,6 +1,6 @@ """ -End-to-end grid v4 regression suite for reasoning_effort mapping across -Anthropic-backed routes. +End-to-end regression suite for reasoning_effort mapping across the +Anthropic-backed routes covered by the original QA sweep. Encodes the 21 (provider x model) x 11 effort matrix (231 cells) from the QA sweep on https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610 @@ -13,9 +13,12 @@ against. Each cell asserts: - Status code returned by LiteLLM (200 vs BadRequestError -> 400) -- the regression signal for clean-error vs leaked-500 mappings. -Hits real provider endpoints. Each route is skipped at runtime when its -required env vars are absent, so PR builds without provider credentials no-op -gracefully. +Calls go to real provider endpoints, but the parent +``tests/llm_translation/conftest.py`` auto-applies ``@pytest.mark.vcr`` to +every collected item, so first run records cassettes (Redis-backed) and +subsequent CI runs replay them with no live spend. Each route still skips at +runtime when its required env vars are absent, so PR builds without provider +credentials no-op gracefully. """ import os @@ -180,7 +183,7 @@ async def _call_messages( @pytest.mark.parametrize( ("route_name", "model", "effort", "cell"), _PARAMS, ids=_PARAM_IDS ) -async def test_reasoning_effort_grid_v4( +async def test_reasoning_effort_grid( route_name: str, model: ModelEntry, effort: str, @@ -209,7 +212,7 @@ async def test_reasoning_effort_grid_v4( raise -def test_grid_v4_cell_count() -> None: +def test_grid_cell_count() -> None: """Guard against accidental drops or duplicates in the grid spec.""" assert len(_PARAMS) == 21 * 11, ( f"expected 231 cells (21 provider x model combos x 11 efforts), " @@ -217,7 +220,7 @@ def test_grid_v4_cell_count() -> None: ) -def test_grid_v4_route_coverage() -> None: +def test_grid_route_coverage() -> None: """The grid must cover every route the original QA sweep covered.""" route_names = {route.name for route in ROUTES} assert route_names == {