mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
refactor(tests): move reasoning_effort grid suite under llm_translation, drop v4 naming
- Drop the "v4" suffix throughout: it referred to the QA sweep iteration,
not this test suite. There's only one regression suite, so just call it
reasoning_effort_grid.
- Move tests/test_litellm/reasoning_effort_grid_v4/ -> tests/llm_translation/
reasoning_effort_grid/. Two reasons:
1. The parent tests/test_litellm/conftest.py installs an autouse fixture
(isolate_host_aws_config) that clears every AWS_* env var before each
test, which would silently skip every Bedrock cell.
2. tests/llm_translation/conftest.py already wires up the Redis-backed
VCR persister and auto-applies @pytest.mark.vcr to every collected
item via apply_vcr_auto_marker_to_items. Living under that conftest
means the suite gets cassette replay for free -- first CI run with
provider creds records 231 cassettes, every subsequent run replays
them with no live spend.
- Trim the suite's own conftest down to just the wire_capture fixture; the
inherited llm_translation conftest covers the VCR plumbing.
- Drop the dedicated reasoning_effort_grid_v4_e2e CircleCI job. The existing
llm_translation_testing job globs tests/llm_translation/**/test_*.py, so
the suite is gated by an existing job with no new wiring.
This commit is contained in:
parent
fec4ae69e0
commit
f77324d766
5 changed files with 31 additions and 77 deletions
|
|
@ -576,48 +576,6 @@ jobs:
|
|||
no_output_timeout: 15m
|
||||
|
||||
# Store test results
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
reasoning_effort_grid_v4_e2e:
|
||||
docker:
|
||||
- *python312_image
|
||||
working_directory: ~/project
|
||||
resource_class: large
|
||||
|
||||
steps:
|
||||
- checkout
|
||||
- setup_google_dns
|
||||
- install_uv
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
uv sync --frozen --all-groups --all-extras --python 3.12
|
||||
- save_cache:
|
||||
paths:
|
||||
- ~/.cache/uv
|
||||
key: v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
# Grid v4 exercises reasoning_effort mapping against real Anthropic,
|
||||
# Azure AI Foundry, Vertex AI, Bedrock Converse, and Bedrock Invoke
|
||||
# endpoints. Per-route cells pytest-skip themselves when the matching
|
||||
# provider env vars are absent, so PRs without credentials no-op.
|
||||
- run:
|
||||
name: Run reasoning_effort grid v4 e2e suite
|
||||
command: |
|
||||
mkdir -p test-results
|
||||
uv run --no-sync python -m pytest \
|
||||
tests/test_litellm/reasoning_effort_grid_v4/ \
|
||||
-v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=20 \
|
||||
-n 4 \
|
||||
--timeout=180 --timeout_method=thread \
|
||||
--retries 2 --retry-delay 5 \
|
||||
--max-worker-restart=5
|
||||
no_output_timeout: 20m
|
||||
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
realtime_translation_testing:
|
||||
|
|
@ -2661,8 +2619,6 @@ workflows:
|
|||
filters: *main_branches
|
||||
- llm_translation_testing:
|
||||
filters: *main_branches
|
||||
- reasoning_effort_grid_v4_e2e:
|
||||
filters: *main_branches
|
||||
- realtime_translation_testing:
|
||||
filters: *main_branches
|
||||
- agent_testing:
|
||||
|
|
|
|||
|
|
@ -1,6 +1,12 @@
|
|||
"""Shared fixtures for the reasoning_effort grid v4 e2e suite."""
|
||||
"""Shared fixtures for the reasoning_effort grid e2e suite.
|
||||
|
||||
VCR wiring (Redis-backed cassette persister, auto-application of
|
||||
``@pytest.mark.vcr`` to every collected item, cassette-cache health summary)
|
||||
is inherited from ``tests/llm_translation/conftest.py``. This file only
|
||||
contributes the ``wire_capture`` fixture, which records the wire body
|
||||
LiteLLM sends upstream so each cell can inspect it.
|
||||
"""
|
||||
|
||||
import os
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import pytest
|
||||
|
|
@ -12,10 +18,10 @@ from litellm.integrations.custom_logger import CustomLogger
|
|||
class _WireBodyCapture(CustomLogger):
|
||||
"""Pre-call hook that records the outgoing wire body LiteLLM sends upstream.
|
||||
|
||||
`complete_input_dict` is the fully transformed provider request as set by
|
||||
every provider transformation in `litellm/llms/**`. Capturing it here means
|
||||
a regression anywhere in the transformation chain (strip, rename, drop)
|
||||
surfaces as an assertion failure on the cell that depends on it.
|
||||
``complete_input_dict`` is the fully transformed provider request as set
|
||||
by every provider transformation in ``litellm/llms/**``. Capturing it here
|
||||
means a regression anywhere in the transformation chain (strip, rename,
|
||||
drop) surfaces as an assertion failure on the cell that depends on it.
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
|
|
@ -37,9 +43,6 @@ class _WireBodyCapture(CustomLogger):
|
|||
def latest(self) -> Optional[Dict[str, Any]]:
|
||||
return self.records[-1] if self.records else None
|
||||
|
||||
def reset(self) -> None:
|
||||
self.records.clear()
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def wire_capture():
|
||||
|
|
@ -50,12 +53,3 @@ def wire_capture():
|
|||
yield capture
|
||||
finally:
|
||||
litellm.callbacks = previous
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def vertex_credentials_path() -> Optional[str]:
|
||||
"""Resolve a usable Vertex credentials file path or None."""
|
||||
path = os.environ.get("GOOGLE_APPLICATION_CREDENTIALS")
|
||||
if path and os.path.exists(path):
|
||||
return path
|
||||
return None
|
||||
|
|
@ -1,12 +1,13 @@
|
|||
"""
|
||||
Canonical post-fix expectations for the reasoning_effort grid v4 sweep.
|
||||
Canonical post-fix expectations for the reasoning_effort grid sweep.
|
||||
|
||||
The QA sweep on https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610
|
||||
covered 21 (provider x model) combos x 11 effort values (231 cells). The follow-up
|
||||
PR https://github.com/BerriAI/litellm/pull/27074 closed nine bugs surfaced by that
|
||||
sweep. This module encodes the post-fix expectations as a small rule set keyed by
|
||||
(model_mode, effort) and per-model capability overrides, then expands them across
|
||||
the model x effort matrix per route.
|
||||
The original QA sweep on
|
||||
https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610
|
||||
covered 21 (provider x model) combos x 11 effort values (231 cells). The
|
||||
follow-up PR https://github.com/BerriAI/litellm/pull/27074 closed nine bugs
|
||||
surfaced by that sweep. This module encodes the post-fix expectations as a
|
||||
small rule set keyed by (model_mode, effort) and per-model capability
|
||||
overrides, then expands them across the model x effort matrix per route.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
|
@ -1,6 +1,6 @@
|
|||
"""
|
||||
End-to-end grid v4 regression suite for reasoning_effort mapping across
|
||||
Anthropic-backed routes.
|
||||
End-to-end regression suite for reasoning_effort mapping across the
|
||||
Anthropic-backed routes covered by the original QA sweep.
|
||||
|
||||
Encodes the 21 (provider x model) x 11 effort matrix (231 cells) from the
|
||||
QA sweep on https://github.com/BerriAI/litellm/pull/27039#issuecomment-4363363610
|
||||
|
|
@ -13,9 +13,12 @@ against. Each cell asserts:
|
|||
- Status code returned by LiteLLM (200 vs BadRequestError -> 400) -- the
|
||||
regression signal for clean-error vs leaked-500 mappings.
|
||||
|
||||
Hits real provider endpoints. Each route is skipped at runtime when its
|
||||
required env vars are absent, so PR builds without provider credentials no-op
|
||||
gracefully.
|
||||
Calls go to real provider endpoints, but the parent
|
||||
``tests/llm_translation/conftest.py`` auto-applies ``@pytest.mark.vcr`` to
|
||||
every collected item, so first run records cassettes (Redis-backed) and
|
||||
subsequent CI runs replay them with no live spend. Each route still skips at
|
||||
runtime when its required env vars are absent, so PR builds without provider
|
||||
credentials no-op gracefully.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
|
@ -180,7 +183,7 @@ async def _call_messages(
|
|||
@pytest.mark.parametrize(
|
||||
("route_name", "model", "effort", "cell"), _PARAMS, ids=_PARAM_IDS
|
||||
)
|
||||
async def test_reasoning_effort_grid_v4(
|
||||
async def test_reasoning_effort_grid(
|
||||
route_name: str,
|
||||
model: ModelEntry,
|
||||
effort: str,
|
||||
|
|
@ -209,7 +212,7 @@ async def test_reasoning_effort_grid_v4(
|
|||
raise
|
||||
|
||||
|
||||
def test_grid_v4_cell_count() -> None:
|
||||
def test_grid_cell_count() -> None:
|
||||
"""Guard against accidental drops or duplicates in the grid spec."""
|
||||
assert len(_PARAMS) == 21 * 11, (
|
||||
f"expected 231 cells (21 provider x model combos x 11 efforts), "
|
||||
|
|
@ -217,7 +220,7 @@ def test_grid_v4_cell_count() -> None:
|
|||
)
|
||||
|
||||
|
||||
def test_grid_v4_route_coverage() -> None:
|
||||
def test_grid_route_coverage() -> None:
|
||||
"""The grid must cover every route the original QA sweep covered."""
|
||||
route_names = {route.name for route in ROUTES}
|
||||
assert route_names == {
|
||||
Loading…
Add table
Reference in a new issue