mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
test(e2e): isolate missing-secret failures to their cells and pin module fixture markers
Register credential and model cleanup as each one succeeds, so a failure partway through deployments setup leaves nothing behind. A provider whose secret is absent now fails only its inline and stored_credential cells by variable name; its os.environ/ cells and every other provider still run. Module-scoped fixtures allocate deterministic markers against the module node, not whichever cell pytest happens to run first, so a full recording replays a narrowed -k or E2E_MATRIX_* selection Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
eb35a070a4
commit
09f01a8445
6 changed files with 55 additions and 67 deletions
|
|
@ -140,7 +140,7 @@ Tests marked `@pytest.mark.e2e` hard-fail when no proxy answers `/health/livelin
|
|||
|
||||
### The endpoint matrix
|
||||
|
||||
`llm_translation/test_endpoint_matrix_e2e.py` runs one scenario per inference endpoint (`/chat/completions`, `/v1/completions`, `/v1/messages`, `/v1/responses`, `/embeddings`, `/v1/images/generations`, `/v1/images/edits`, `/v1/audio/speech`, `/v1/audio/transcriptions`, `/v1/moderations`, streamed and not where the endpoint streams) over every provider in `llm_translation/endpoint_matrix.py` that serves it, and over every way the proxy can hold that provider's secret: `os.environ/` references, literal values, and a stored `litellm_credential_name`. Each cell is one pytest parameter carrying its exact registry id, so a translation bug fixed on `/chat/completions` but not on `/v1/messages` shows up as one red cell next to green ones instead of a test gap. Adding a provider is one `Provider` entry in the catalog naming its route, credential environment variables, and backend model per endpoint; it needs no new test code. `E2E_MATRIX_PROVIDERS` and `E2E_MATRIX_AUTH_MODES` narrow a run to a comma-separated subset of catalog routes and auth modes, unknown names fail at collection, and a provider whose secret is missing from the runner fails its inline and stored-credential cells by variable name rather than skipping
|
||||
`llm_translation/test_endpoint_matrix_e2e.py` runs one scenario per inference endpoint (`/chat/completions`, `/v1/completions`, `/v1/messages`, `/v1/responses`, `/embeddings`, `/v1/images/generations`, `/v1/images/edits`, `/v1/audio/speech`, `/v1/audio/transcriptions`, `/v1/moderations`, streamed and not where the endpoint streams) over every provider in `llm_translation/endpoint_matrix.py` that serves it, and over every way the proxy can hold that provider's secret: `os.environ/` references, literal values, and a stored `litellm_credential_name`. Each cell is one pytest parameter carrying its exact registry id, so a translation bug fixed on `/chat/completions` but not on `/v1/messages` shows up as one red cell next to green ones instead of a test gap. Adding a provider is one `Provider` entry in the catalog naming its route, credential environment variables, and backend model per endpoint; it needs no new test code. `E2E_MATRIX_PROVIDERS` and `E2E_MATRIX_AUTH_MODES` narrow a run to a comma-separated subset of catalog routes and auth modes, unknown names fail at collection, and a provider whose secret is missing from the runner fails its inline and stored-credential cells by variable name rather than skipping, while its `os.environ/` cells and every other provider still run
|
||||
|
||||
```bash
|
||||
E2E_MATRIX_PROVIDERS=openai,anthropic E2E_MATRIX_AUTH_MODES=env_ref uv run pytest tests/e2e/llm_translation/test_endpoint_matrix_e2e.py
|
||||
|
|
|
|||
|
|
@ -95,12 +95,17 @@ class ReplayMiss(AssertionError):
|
|||
_marker_ordinals: Final[dict[str, int]] = {}
|
||||
|
||||
|
||||
def marker_owner() -> str:
|
||||
owner = registration_owner()
|
||||
return current_test_key() if owner == SESSION_TEST_KEY else owner
|
||||
|
||||
|
||||
def deterministic_marker() -> str:
|
||||
"""Stable stand-in for uuid-based unique markers in record and replay modes:
|
||||
the Nth marker of a test is a pure function of the test's node id and N, so a
|
||||
replay run regenerates exactly the model names, prompts, and tags the record
|
||||
run sent and every recorded provider interaction still matches its key."""
|
||||
test_key = current_test_key()
|
||||
the Nth marker of a node is a pure function of its id and N, so a replay run
|
||||
regenerates exactly the model names, prompts, and tags the record run sent and
|
||||
every recorded provider interaction still matches its key."""
|
||||
test_key = marker_owner()
|
||||
ordinal = _marker_ordinals.get(test_key, 0)
|
||||
_marker_ordinals[test_key] = ordinal + 1
|
||||
return hashlib.sha1(f"{test_key}#{ordinal}".encode()).hexdigest()[:12]
|
||||
|
|
|
|||
|
|
@ -1,14 +1,4 @@
|
|||
"""Provider catalog behind test_endpoint_matrix_e2e.py.
|
||||
|
||||
One `Provider` row per route the matrix drives. Adding a provider is adding a row
|
||||
here (its credential fields, its backend model per endpoint family, and the edge
|
||||
mount when the provider edge can record and replay it) plus the registry cells
|
||||
those combinations claim. The test module never names a provider.
|
||||
|
||||
`E2E_MATRIX_PROVIDERS` and `E2E_MATRIX_AUTH_MODES` narrow the default of every
|
||||
catalog row and every auth mode; an unknown name fails collection instead of
|
||||
quietly selecting nothing.
|
||||
"""
|
||||
"""Provider catalog behind test_endpoint_matrix_e2e.py: one row per route, selectable by env var."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
|
@ -174,12 +164,12 @@ def _selection(variable: str, known: tuple[str, ...]) -> tuple[str, ...]:
|
|||
return chosen
|
||||
|
||||
|
||||
def missing_credentials(provider: Provider) -> tuple[str, ...]:
|
||||
return tuple(env for env in provider.credential.values() if not os.environ.get(env))
|
||||
|
||||
|
||||
def credential_values(provider: Provider) -> Mapping[str, str]:
|
||||
"""The provider's real secrets as read from this process's environment, for the
|
||||
auth modes that hand the proxy literal values instead of `os.environ/` references.
|
||||
Missing variables fail by name rather than registering a deployment that will
|
||||
fail later with a less specific provider error."""
|
||||
missing: Final = tuple(env for env in provider.credential.values() if not os.environ.get(env))
|
||||
missing: Final = missing_credentials(provider)
|
||||
if missing:
|
||||
raise RuntimeError(f"{provider.route} inline/stored auth needs {', '.join(missing)} in the test environment")
|
||||
return MappingProxyType({field: os.environ[env] for field, env in provider.credential.items()})
|
||||
|
|
|
|||
|
|
@ -1,29 +1,10 @@
|
|||
"""One scenario per non-management inference endpoint, run over every catalog
|
||||
provider that serves it and every way the proxy can hold that provider's secret.
|
||||
|
||||
A cell is (endpoint case, provider, auth mode). The case knows how to call the
|
||||
endpoint and what a meaningful answer looks like; the provider knows its backend
|
||||
model and credential fields; the auth mode decides whether the deployment carries
|
||||
`os.environ/` references, literal values, or a `litellm_credential_name`. The
|
||||
same scenario therefore hits `/v1/messages`, `/v1/responses`, `/chat/completions`
|
||||
and the rest identically, so a translation bug fixed on one endpoint but not
|
||||
another shows up as one red cell next to green ones.
|
||||
|
||||
Every cell calls the proxy the way a customer does: the OpenAI SDK for the
|
||||
OpenAI-compatible surface and the Anthropic SDK for `/v1/messages`, each holding
|
||||
a fresh virtual key. OpenAI and Anthropic cells are edge-wired and replayable;
|
||||
every other provider runs live in every fixture mode. Streaming cases only assert
|
||||
grammar and content, never provider timing, so replay stays fast.
|
||||
|
||||
The selected cells' deployments are registered as one batch before the first cell
|
||||
runs, so the module pays the data-plane reload budget once instead of once per
|
||||
cell; each cell still calls through its own fresh virtual key.
|
||||
"""
|
||||
"""Live e2e: one scenario per inference endpoint over every catalog provider and auth mode."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
from collections.abc import Callable, Iterator, Mapping
|
||||
from contextlib import ExitStack
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from types import MappingProxyType
|
||||
|
|
@ -41,6 +22,7 @@ from endpoint_matrix import (
|
|||
Streaming,
|
||||
credential_values,
|
||||
deployment_params,
|
||||
missing_credentials,
|
||||
selected_auth_modes,
|
||||
selected_providers,
|
||||
)
|
||||
|
|
@ -291,8 +273,6 @@ def _param(cell: MatrixCell) -> ParameterSet:
|
|||
|
||||
|
||||
def _selected_cells(session: pytest.Session) -> tuple[MatrixCell, ...]:
|
||||
"""The cells pytest will actually run from this module, after -m / -k deselection,
|
||||
so replay registers no deployment for a provider it holds no credential for."""
|
||||
return tuple(
|
||||
cell
|
||||
for item in session.items
|
||||
|
|
@ -314,12 +294,13 @@ def _deployment_body(key: DeploymentKey, provider: Provider, credential_name: st
|
|||
)
|
||||
|
||||
|
||||
def _registrable(cell: MatrixCell) -> bool:
|
||||
return cell.auth_mode == "env_ref" or not missing_credentials(cell.provider)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def deployments(request: pytest.FixtureRequest, proxy: ProxyClient) -> Iterator[Mapping[DeploymentKey, str]]:
|
||||
"""One deployment per (provider, endpoint, auth mode) among the selected cells, written
|
||||
as a single batch, plus one stored credential per provider that any selected cell
|
||||
references by name. Yields deployment key -> model alias; tears everything down."""
|
||||
cells: Final = _selected_cells(request.session)
|
||||
cells: Final = tuple(cell for cell in _selected_cells(request.session) if _registrable(cell))
|
||||
providers: Final[Mapping[LlmRoute, Provider]] = MappingProxyType(
|
||||
{cell.provider.route: cell.provider for cell in cells}
|
||||
)
|
||||
|
|
@ -329,25 +310,20 @@ def deployments(request: pytest.FixtureRequest, proxy: ProxyClient) -> Iterator[
|
|||
credentials: Final[Mapping[LlmRoute, str]] = MappingProxyType(
|
||||
{route: f"e2e-matrix-cred-{unique_marker()}" for route in stored}
|
||||
)
|
||||
for route, name in credentials.items():
|
||||
proxy.create_credential(
|
||||
CredentialCreateBody(credential_name=name, credential_values=dict(credential_values(providers[route])))
|
||||
)
|
||||
try:
|
||||
with ExitStack() as teardown:
|
||||
for route, name in credentials.items():
|
||||
proxy.create_credential(
|
||||
CredentialCreateBody(credential_name=name, credential_values=dict(credential_values(providers[route])))
|
||||
)
|
||||
_ = teardown.callback(proxy.delete_credential, name)
|
||||
keys: Final = tuple(dict.fromkeys(cell.deployment for cell in cells))
|
||||
bodies: Final = tuple(
|
||||
_deployment_body(key, providers[key[0]], credentials[key[0]] if key[2] == "stored_credential" else None)
|
||||
for key in keys
|
||||
)
|
||||
model_ids: Final = proxy.register_models(bodies)
|
||||
try:
|
||||
yield MappingProxyType(dict(zip(keys, (body.model_name for body in bodies), strict=True)))
|
||||
finally:
|
||||
for model_id in model_ids:
|
||||
proxy.delete_model(model_id)
|
||||
finally:
|
||||
for name in credentials.values():
|
||||
proxy.delete_credential(name)
|
||||
for model_id in proxy.register_models(bodies):
|
||||
_ = teardown.callback(proxy.delete_model, model_id)
|
||||
yield MappingProxyType(dict(zip(keys, (body.model_name for body in bodies), strict=True)))
|
||||
|
||||
|
||||
class TestEndpointMatrix:
|
||||
|
|
@ -359,4 +335,9 @@ class TestEndpointMatrix:
|
|||
resources: ResourceManager,
|
||||
deployments: Mapping[DeploymentKey, str],
|
||||
) -> None:
|
||||
cell.case.run(sdk, resources.key(), deployments[cell.deployment])
|
||||
model: Final = deployments.get(cell.deployment)
|
||||
assert model is not None, (
|
||||
f"{cell.provider.route} {cell.auth_mode} auth needs "
|
||||
f"{', '.join(missing_credentials(cell.provider))} in the test environment"
|
||||
)
|
||||
cell.case.run(sdk, resources.key(), model)
|
||||
|
|
|
|||
|
|
@ -665,10 +665,7 @@ class ProxyClient:
|
|||
return model_id
|
||||
|
||||
def register_models(self, bodies: Sequence[ModelNewBody]) -> tuple[str, ...]:
|
||||
"""`register_model` for a batch of proxy-wide deployments: every row is written
|
||||
first, then each is awaited on the data plane, and one propagation wait covers
|
||||
them all, so a suite registering many deployments pays the reload budget once.
|
||||
A failure anywhere deletes every deployment the batch already created."""
|
||||
"""`register_model` for a batch, paying the propagation wait once; a failure deletes the whole batch."""
|
||||
results: Final = tuple(self._write_model(body) for body in bodies)
|
||||
written_at = time.monotonic()
|
||||
model_ids: Final = tuple(result.data.model_id for result in results if isinstance(result, Success))
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from __future__ import annotations
|
|||
import hashlib
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -28,6 +29,13 @@ from fixture_mode import (
|
|||
NOW = datetime(2026, 8, 18, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def module_marker(request: pytest.FixtureRequest) -> tuple[str, str]:
|
||||
node: Final = request.node # pyright: ignore[reportUnknownMemberType, reportUnknownVariableType] # pytest: untyped
|
||||
assert isinstance(node, pytest.Module)
|
||||
return node.nodeid, deterministic_marker()
|
||||
|
||||
|
||||
def write_manifest(root: Path, recorded_at: datetime) -> None:
|
||||
root.mkdir(parents=True, exist_ok=True)
|
||||
manifest = Manifest(
|
||||
|
|
@ -57,6 +65,13 @@ class TestDeterministicMarker:
|
|||
assert deterministic_marker() == hashlib.sha1(f"{key}#0".encode()).hexdigest()[:12]
|
||||
assert deterministic_marker() == hashlib.sha1(f"{key}#1".encode()).hexdigest()[:12]
|
||||
|
||||
def test_module_fixture_markers_belong_to_the_module_not_its_first_test(
|
||||
self, module_marker: tuple[str, str]
|
||||
) -> None:
|
||||
module_id, marker = module_marker
|
||||
assert module_id != current_test_key()
|
||||
assert marker == hashlib.sha1(f"{module_id}#0".encode()).hexdigest()[:12]
|
||||
|
||||
|
||||
class TestCurrentTestKey:
|
||||
def test_names_this_test_and_strips_the_phase(self) -> None:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue