diff --git a/tests/e2e/CONTRIBUTING.md b/tests/e2e/CONTRIBUTING.md index 0b4e4e7e6c4..370d93fefbe 100644 --- a/tests/e2e/CONTRIBUTING.md +++ b/tests/e2e/CONTRIBUTING.md @@ -140,7 +140,7 @@ Tests marked `@pytest.mark.e2e` hard-fail when no proxy answers `/health/livelin ### The endpoint matrix -`llm_translation/test_endpoint_matrix_e2e.py` runs one scenario per inference endpoint (`/chat/completions`, `/v1/completions`, `/v1/messages`, `/v1/responses`, `/embeddings`, `/v1/images/generations`, `/v1/images/edits`, `/v1/audio/speech`, `/v1/audio/transcriptions`, `/v1/moderations`, streamed and not where the endpoint streams) over every provider in `llm_translation/endpoint_matrix.py` that serves it, and over every way the proxy can hold that provider's secret: `os.environ/` references, literal values, and a stored `litellm_credential_name`. Each cell is one pytest parameter carrying its exact registry id, so a translation bug fixed on `/chat/completions` but not on `/v1/messages` shows up as one red cell next to green ones instead of a test gap. Adding a provider is one `Provider` entry in the catalog naming its route, credential environment variables, and backend model per endpoint; it needs no new test code. `E2E_MATRIX_PROVIDERS` and `E2E_MATRIX_AUTH_MODES` narrow a run to a comma-separated subset of catalog routes and auth modes, unknown names fail at collection, and a provider whose secret is missing from the runner fails its inline and stored-credential cells by variable name rather than skipping +`llm_translation/test_endpoint_matrix_e2e.py` runs one scenario per inference endpoint (`/chat/completions`, `/v1/completions`, `/v1/messages`, `/v1/responses`, `/embeddings`, `/v1/images/generations`, `/v1/images/edits`, `/v1/audio/speech`, `/v1/audio/transcriptions`, `/v1/moderations`, streamed and not where the endpoint streams) over every provider in `llm_translation/endpoint_matrix.py` that serves it, and over every way the proxy can hold that provider's secret: `os.environ/` references, literal values, and a stored `litellm_credential_name`. Each cell is one pytest parameter carrying its exact registry id, so a translation bug fixed on `/chat/completions` but not on `/v1/messages` shows up as one red cell next to green ones instead of a test gap. Adding a provider is one `Provider` entry in the catalog naming its route, credential environment variables, and backend model per endpoint; it needs no new test code. `E2E_MATRIX_PROVIDERS` and `E2E_MATRIX_AUTH_MODES` narrow a run to a comma-separated subset of catalog routes and auth modes, unknown names fail at collection, and a provider whose secret is missing from the runner fails its inline and stored-credential cells by variable name rather than skipping, while its `os.environ/` cells and every other provider still run ```bash E2E_MATRIX_PROVIDERS=openai,anthropic E2E_MATRIX_AUTH_MODES=env_ref uv run pytest tests/e2e/llm_translation/test_endpoint_matrix_e2e.py diff --git a/tests/e2e/fixture_mode.py b/tests/e2e/fixture_mode.py index 26b315c0c06..e6f314ba9d5 100644 --- a/tests/e2e/fixture_mode.py +++ b/tests/e2e/fixture_mode.py @@ -95,12 +95,17 @@ class ReplayMiss(AssertionError): _marker_ordinals: Final[dict[str, int]] = {} +def marker_owner() -> str: + owner = registration_owner() + return current_test_key() if owner == SESSION_TEST_KEY else owner + + def deterministic_marker() -> str: """Stable stand-in for uuid-based unique markers in record and replay modes: - the Nth marker of a test is a pure function of the test's node id and N, so a - replay run regenerates exactly the model names, prompts, and tags the record - run sent and every recorded provider interaction still matches its key.""" - test_key = current_test_key() + the Nth marker of a node is a pure function of its id and N, so a replay run + regenerates exactly the model names, prompts, and tags the record run sent and + every recorded provider interaction still matches its key.""" + test_key = marker_owner() ordinal = _marker_ordinals.get(test_key, 0) _marker_ordinals[test_key] = ordinal + 1 return hashlib.sha1(f"{test_key}#{ordinal}".encode()).hexdigest()[:12] diff --git a/tests/e2e/llm_translation/endpoint_matrix.py b/tests/e2e/llm_translation/endpoint_matrix.py index e391e9f3078..da54089cab8 100644 --- a/tests/e2e/llm_translation/endpoint_matrix.py +++ b/tests/e2e/llm_translation/endpoint_matrix.py @@ -1,14 +1,4 @@ -"""Provider catalog behind test_endpoint_matrix_e2e.py. - -One `Provider` row per route the matrix drives. Adding a provider is adding a row -here (its credential fields, its backend model per endpoint family, and the edge -mount when the provider edge can record and replay it) plus the registry cells -those combinations claim. The test module never names a provider. - -`E2E_MATRIX_PROVIDERS` and `E2E_MATRIX_AUTH_MODES` narrow the default of every -catalog row and every auth mode; an unknown name fails collection instead of -quietly selecting nothing. -""" +"""Provider catalog behind test_endpoint_matrix_e2e.py: one row per route, selectable by env var.""" from __future__ import annotations @@ -174,12 +164,12 @@ def _selection(variable: str, known: tuple[str, ...]) -> tuple[str, ...]: return chosen +def missing_credentials(provider: Provider) -> tuple[str, ...]: + return tuple(env for env in provider.credential.values() if not os.environ.get(env)) + + def credential_values(provider: Provider) -> Mapping[str, str]: - """The provider's real secrets as read from this process's environment, for the - auth modes that hand the proxy literal values instead of `os.environ/` references. - Missing variables fail by name rather than registering a deployment that will - fail later with a less specific provider error.""" - missing: Final = tuple(env for env in provider.credential.values() if not os.environ.get(env)) + missing: Final = missing_credentials(provider) if missing: raise RuntimeError(f"{provider.route} inline/stored auth needs {', '.join(missing)} in the test environment") return MappingProxyType({field: os.environ[env] for field, env in provider.credential.items()}) diff --git a/tests/e2e/llm_translation/test_endpoint_matrix_e2e.py b/tests/e2e/llm_translation/test_endpoint_matrix_e2e.py index 4dadd8fd8fe..1f577a0c46a 100644 --- a/tests/e2e/llm_translation/test_endpoint_matrix_e2e.py +++ b/tests/e2e/llm_translation/test_endpoint_matrix_e2e.py @@ -1,29 +1,10 @@ -"""One scenario per non-management inference endpoint, run over every catalog -provider that serves it and every way the proxy can hold that provider's secret. - -A cell is (endpoint case, provider, auth mode). The case knows how to call the -endpoint and what a meaningful answer looks like; the provider knows its backend -model and credential fields; the auth mode decides whether the deployment carries -`os.environ/` references, literal values, or a `litellm_credential_name`. The -same scenario therefore hits `/v1/messages`, `/v1/responses`, `/chat/completions` -and the rest identically, so a translation bug fixed on one endpoint but not -another shows up as one red cell next to green ones. - -Every cell calls the proxy the way a customer does: the OpenAI SDK for the -OpenAI-compatible surface and the Anthropic SDK for `/v1/messages`, each holding -a fresh virtual key. OpenAI and Anthropic cells are edge-wired and replayable; -every other provider runs live in every fixture mode. Streaming cases only assert -grammar and content, never provider timing, so replay stays fast. - -The selected cells' deployments are registered as one batch before the first cell -runs, so the module pays the data-plane reload budget once instead of once per -cell; each cell still calls through its own fresh virtual key. -""" +"""Live e2e: one scenario per inference endpoint over every catalog provider and auth mode.""" from __future__ import annotations import base64 from collections.abc import Callable, Iterator, Mapping +from contextlib import ExitStack from dataclasses import dataclass from pathlib import Path from types import MappingProxyType @@ -41,6 +22,7 @@ from endpoint_matrix import ( Streaming, credential_values, deployment_params, + missing_credentials, selected_auth_modes, selected_providers, ) @@ -291,8 +273,6 @@ def _param(cell: MatrixCell) -> ParameterSet: def _selected_cells(session: pytest.Session) -> tuple[MatrixCell, ...]: - """The cells pytest will actually run from this module, after -m / -k deselection, - so replay registers no deployment for a provider it holds no credential for.""" return tuple( cell for item in session.items @@ -314,12 +294,13 @@ def _deployment_body(key: DeploymentKey, provider: Provider, credential_name: st ) +def _registrable(cell: MatrixCell) -> bool: + return cell.auth_mode == "env_ref" or not missing_credentials(cell.provider) + + @pytest.fixture(scope="module") def deployments(request: pytest.FixtureRequest, proxy: ProxyClient) -> Iterator[Mapping[DeploymentKey, str]]: - """One deployment per (provider, endpoint, auth mode) among the selected cells, written - as a single batch, plus one stored credential per provider that any selected cell - references by name. Yields deployment key -> model alias; tears everything down.""" - cells: Final = _selected_cells(request.session) + cells: Final = tuple(cell for cell in _selected_cells(request.session) if _registrable(cell)) providers: Final[Mapping[LlmRoute, Provider]] = MappingProxyType( {cell.provider.route: cell.provider for cell in cells} ) @@ -329,25 +310,20 @@ def deployments(request: pytest.FixtureRequest, proxy: ProxyClient) -> Iterator[ credentials: Final[Mapping[LlmRoute, str]] = MappingProxyType( {route: f"e2e-matrix-cred-{unique_marker()}" for route in stored} ) - for route, name in credentials.items(): - proxy.create_credential( - CredentialCreateBody(credential_name=name, credential_values=dict(credential_values(providers[route]))) - ) - try: + with ExitStack() as teardown: + for route, name in credentials.items(): + proxy.create_credential( + CredentialCreateBody(credential_name=name, credential_values=dict(credential_values(providers[route]))) + ) + _ = teardown.callback(proxy.delete_credential, name) keys: Final = tuple(dict.fromkeys(cell.deployment for cell in cells)) bodies: Final = tuple( _deployment_body(key, providers[key[0]], credentials[key[0]] if key[2] == "stored_credential" else None) for key in keys ) - model_ids: Final = proxy.register_models(bodies) - try: - yield MappingProxyType(dict(zip(keys, (body.model_name for body in bodies), strict=True))) - finally: - for model_id in model_ids: - proxy.delete_model(model_id) - finally: - for name in credentials.values(): - proxy.delete_credential(name) + for model_id in proxy.register_models(bodies): + _ = teardown.callback(proxy.delete_model, model_id) + yield MappingProxyType(dict(zip(keys, (body.model_name for body in bodies), strict=True))) class TestEndpointMatrix: @@ -359,4 +335,9 @@ class TestEndpointMatrix: resources: ResourceManager, deployments: Mapping[DeploymentKey, str], ) -> None: - cell.case.run(sdk, resources.key(), deployments[cell.deployment]) + model: Final = deployments.get(cell.deployment) + assert model is not None, ( + f"{cell.provider.route} {cell.auth_mode} auth needs " + f"{', '.join(missing_credentials(cell.provider))} in the test environment" + ) + cell.case.run(sdk, resources.key(), model) diff --git a/tests/e2e/proxy_client.py b/tests/e2e/proxy_client.py index 75d755578f1..02a6b28467a 100644 --- a/tests/e2e/proxy_client.py +++ b/tests/e2e/proxy_client.py @@ -665,10 +665,7 @@ class ProxyClient: return model_id def register_models(self, bodies: Sequence[ModelNewBody]) -> tuple[str, ...]: - """`register_model` for a batch of proxy-wide deployments: every row is written - first, then each is awaited on the data plane, and one propagation wait covers - them all, so a suite registering many deployments pays the reload budget once. - A failure anywhere deletes every deployment the batch already created.""" + """`register_model` for a batch, paying the propagation wait once; a failure deletes the whole batch.""" results: Final = tuple(self._write_model(body) for body in bodies) written_at = time.monotonic() model_ids: Final = tuple(result.data.model_id for result in results if isinstance(result, Success)) diff --git a/tests/e2e/test_fixture_mode.py b/tests/e2e/test_fixture_mode.py index 109bb9e1b11..5e114c9f431 100644 --- a/tests/e2e/test_fixture_mode.py +++ b/tests/e2e/test_fixture_mode.py @@ -12,6 +12,7 @@ from __future__ import annotations import hashlib from datetime import datetime, timedelta, timezone from pathlib import Path +from typing import Final import pytest @@ -28,6 +29,13 @@ from fixture_mode import ( NOW = datetime(2026, 8, 18, 12, 0, 0, tzinfo=timezone.utc) +@pytest.fixture(scope="module") +def module_marker(request: pytest.FixtureRequest) -> tuple[str, str]: + node: Final = request.node # pyright: ignore[reportUnknownMemberType, reportUnknownVariableType] # pytest: untyped + assert isinstance(node, pytest.Module) + return node.nodeid, deterministic_marker() + + def write_manifest(root: Path, recorded_at: datetime) -> None: root.mkdir(parents=True, exist_ok=True) manifest = Manifest( @@ -57,6 +65,13 @@ class TestDeterministicMarker: assert deterministic_marker() == hashlib.sha1(f"{key}#0".encode()).hexdigest()[:12] assert deterministic_marker() == hashlib.sha1(f"{key}#1".encode()).hexdigest()[:12] + def test_module_fixture_markers_belong_to_the_module_not_its_first_test( + self, module_marker: tuple[str, str] + ) -> None: + module_id, marker = module_marker + assert module_id != current_test_key() + assert marker == hashlib.sha1(f"{module_id}#0".encode()).hexdigest()[:12] + class TestCurrentTestKey: def test_names_this_test_and_strips_the_phase(self) -> None: