From 736da28ac15c6ef336cc35f80f56994064803856 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 7 Oct 2026 10:30:28 -0700 Subject: [PATCH 01/13] fix(tests): match the lowercased bind error in the owned-proxy port-race retry (#45097) Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- tests/integration/_support/process.py | 4 +-- .../unit/integration_support/test_process.py | 35 +++++++++++++++---- 2 files changed, 31 insertions(+), 8 deletions(-) diff --git a/tests/integration/_support/process.py b/tests/integration/_support/process.py index 8e9f689ec2d..12780540380 100644 --- a/tests/integration/_support/process.py +++ b/tests/integration/_support/process.py @@ -119,7 +119,7 @@ def _stop(process: subprocess.Popen[bytes]) -> None: _PORT_ATTEMPTS: Final = 3 -_BIND_COLLISION: Final = os.strerror(errno.EADDRINUSE) +_BIND_COLLISION: Final = os.strerror(errno.EADDRINUSE).lower() def _free_port() -> int: @@ -151,7 +151,7 @@ def _launch(command: tuple[str, ...], root: Path, environment: Mapping[str, str] def _lost_port_race(exit_code: int | None, log: Path) -> bool: - return exit_code is not None and _BIND_COLLISION in log.read_text() + return exit_code is not None and _BIND_COLLISION in log.read_text().lower() def _wait_until_ready(launch: _Launch) -> None: diff --git a/tests/unit/integration_support/test_process.py b/tests/unit/integration_support/test_process.py index 042a2447dff..7896786fce8 100644 --- a/tests/unit/integration_support/test_process.py +++ b/tests/unit/integration_support/test_process.py @@ -1,8 +1,10 @@ from __future__ import annotations +import asyncio import errno import importlib -import os +import socket +from collections.abc import Iterator from pathlib import Path from types import ModuleType from typing import Final @@ -10,7 +12,6 @@ from typing import Final import pytest TESTS_DIR: Final = Path(__file__).resolve().parents[2] -BIND_ERROR_LINE: Final = f"ERROR: {OSError(errno.EADDRINUSE, os.strerror(errno.EADDRINUSE))}\n" UNRELATED_CRASH: Final = "Traceback (most recent call last):\nModuleNotFoundError: No module named 'litellm'\n" @@ -20,19 +21,41 @@ def process_module(monkeypatch: pytest.MonkeyPatch) -> ModuleType: return importlib.import_module("integration._support.process") +async def _refused_bind(port: int) -> OSError | None: + try: + await asyncio.get_running_loop().create_server(asyncio.Protocol, "127.0.0.1", port) + except OSError as refused: + return refused + return None + + +@pytest.fixture +def bind_error_line() -> Iterator[str]: + with socket.socket() as held: + held.bind(("127.0.0.1", 0)) + held.listen() + refused: Final = asyncio.run(_refused_bind(held.getsockname()[1])) + assert refused is not None and refused.errno == errno.EADDRINUSE + yield f"ERROR: {refused}\n" + + def _written_log(directory: Path, text: str) -> Path: log: Final = directory / "owned-proxy.log" log.write_text(text) return log -def test_lost_port_race_matches_the_bind_error_the_server_logs(process_module: ModuleType, tmp_path: Path) -> None: - assert process_module._lost_port_race(1, _written_log(tmp_path, BIND_ERROR_LINE)) +def test_lost_port_race_matches_the_bind_error_the_server_logs( + process_module: ModuleType, bind_error_line: str, tmp_path: Path +) -> None: + assert process_module._lost_port_race(1, _written_log(tmp_path, bind_error_line)) def test_lost_port_race_ignores_an_exit_for_another_reason(process_module: ModuleType, tmp_path: Path) -> None: assert not process_module._lost_port_race(1, _written_log(tmp_path, UNRELATED_CRASH)) -def test_lost_port_race_needs_the_process_to_have_exited(process_module: ModuleType, tmp_path: Path) -> None: - assert not process_module._lost_port_race(None, _written_log(tmp_path, BIND_ERROR_LINE)) +def test_lost_port_race_needs_the_process_to_have_exited( + process_module: ModuleType, bind_error_line: str, tmp_path: Path +) -> None: + assert not process_module._lost_port_race(None, _written_log(tmp_path, bind_error_line)) From ade17902a03458b68d23dbcb40bab9b67e389d8e Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Wed, 7 Oct 2026 10:30:37 -0700 Subject: [PATCH 02/13] test(e2e): tag llm_translation tests with Subject metadata and record harness steps (#44950) * test(e2e): add enum values, auto-discovering label gates and secret hiding for e2e metadata * test(e2e): tag llm_translation tests with Subject metadata and record harness steps * docs(e2e): name every markerless harness test file that carries no Subject * test(e2e): keep the step discovery comprehensions to one for clause * test(e2e): declare the realtime param tuples Final --- .../llm_translation/conversational_matrix.py | 66 +++- .../e2e/llm_translation/passthrough_client.py | 18 +- .../realtime/realtime_client.py | 6 +- .../realtime/test_realtime_bedrock_e2e.py | 10 + .../realtime/test_realtime_e2e.py | 55 ++- .../test_realtime_pipecat_audio_e2e.py | 62 ++- .../realtime/test_realtime_pipecat_e2e.py | 17 +- .../llm_translation/test_audio_speech_e2e.py | 70 +++- .../test_audio_transcriptions_e2e.py | 49 ++- .../test_bedrock_native_e2e.py | 90 +++++ .../test_bedrock_provider_matrix_e2e.py | 55 +++ ...test_bedrock_web_search_server_tool_e2e.py | 11 + .../e2e/llm_translation/test_cache_control.py | 41 ++ ..._cache_control_injection_tool_calls_e2e.py | 32 +- .../test_chat_completions_contract_e2e.py | 79 ++++ .../test_chat_completions_regression_e2e.py | 369 +++++++++++++++++- .../test_chat_mid_conversation_system_e2e.py | 41 ++ .../test_chat_stream_contract_e2e.py | 14 +- .../test_chat_tool_round_trip_e2e.py | 51 +++ .../test_completions_endpoint_e2e.py | 16 +- .../llm_translation/test_containers_e2e.py | 26 ++ .../test_credential_messages_e2e.py | 15 +- .../test_custom_pricing_e2e.py | 22 ++ .../test_deepseek_reasoning_e2e.py | 31 ++ .../test_embeddings_endpoint_e2e.py | 108 ++++- .../test_files_batches_contract_e2e.py | 23 ++ .../llm_translation/test_google_native_e2e.py | 19 + .../llm_translation/test_image_edits_e2e.py | 33 +- .../test_image_generation_e2e.py | 61 ++- .../test_messages_azure_foundry_e2e.py | 49 +++ .../test_messages_bedrock_e2e.py | 20 + .../e2e/llm_translation/test_messages_e2e.py | 135 ++++++- ...st_messages_mid_conversation_system_e2e.py | 21 + ...onversation_system_native_providers_e2e.py | 41 ++ .../llm_translation/test_moderations_e2e.py | 32 +- .../e2e/llm_translation/test_ocr_rust_e2e.py | 68 +++- .../llm_translation/test_passthrough_e2e.py | 157 +++++++- .../test_passthrough_headers_e2e.py | 10 + .../test_provider_features_e2e.py | 15 +- .../llm_translation/test_realtime_http_e2e.py | 11 + tests/e2e/llm_translation/test_rerank_e2e.py | 27 +- .../test_responses_bridge_streaming_e2e.py | 31 +- .../e2e/llm_translation/test_responses_e2e.py | 208 +++++++++- .../test_responses_retrieve_e2e.py | 55 ++- tests/e2e/llm_translation/test_sail_e2e.py | 37 ++ .../llm_translation/test_together_ai_e2e.py | 129 ++++++ .../test_token_counter_gemini_contents_e2e.py | 30 +- .../llm_translation/test_vector_stores_e2e.py | 6 + .../test_vertex_passthrough_e2e.py | 10 + 49 files changed, 2489 insertions(+), 93 deletions(-) diff --git a/tests/e2e/llm_translation/conversational_matrix.py b/tests/e2e/llm_translation/conversational_matrix.py index 0d6f6ed3d4e..20dbe236fae 100644 --- a/tests/e2e/llm_translation/conversational_matrix.py +++ b/tests/e2e/llm_translation/conversational_matrix.py @@ -33,6 +33,8 @@ from anthropic.types import ( ToolUseBlockParam, ) from e2e_config import provider_edge_base, unique_marker +from e2e_metadata import Capability as MetaCapability +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta, step from lifecycle import ResourceManager from llm_translation.sdk_clients import NO_PROXY_CACHE, SdkClients, response_header from models import CredentialCreateBody, LiteLLMParamsBody @@ -65,6 +67,10 @@ Streaming = Literal["stream", "nonstream"] Assertion = Literal["works", "cost_logged"] ToolMode = Literal["none", "forced", "offered"] +GPT_4O_MINI_BACKEND: Final = "openai/gpt-4o-mini" +GPT_5_4_MINI_BACKEND: Final = "openai/gpt-5.4-mini" +CLAUDE_HAIKU_BACKEND: Final = "anthropic/claude-haiku-4-5" + SURFACES: Final[tuple[SurfaceName, ...]] = ("chat_completions", "messages", "responses") AUTH_METHODS: Final[tuple[AuthMethod, ...]] = ("env_ref", "stored_credential") @@ -104,12 +110,19 @@ class Deployment: assert key, f"{self.api_key_env} is not set in the test process environment" return key + def provider(self) -> Provider: + match self.route: + case "openai": + return Provider.OPENAI + case "anthropic": + return Provider.ANTHROPIC + DEPLOYMENTS: Final[tuple[Deployment, ...]] = ( Deployment( route="openai", label="gpt-4o-mini", - backend="openai/gpt-4o-mini", + backend=GPT_4O_MINI_BACKEND, api_key_env="OPENAI_API_KEY", edge_mount="openai", edge_suffix="/v1", @@ -117,7 +130,7 @@ DEPLOYMENTS: Final[tuple[Deployment, ...]] = ( Deployment( route="openai", label="gpt-5.4-mini", - backend="openai/gpt-5.4-mini", + backend=GPT_5_4_MINI_BACKEND, api_key_env="OPENAI_API_KEY", edge_mount="openai", edge_suffix="/v1", @@ -125,7 +138,7 @@ DEPLOYMENTS: Final[tuple[Deployment, ...]] = ( Deployment( route="anthropic", label="claude-haiku-4-5", - backend="anthropic/claude-haiku-4-5", + backend=CLAUDE_HAIKU_BACKEND, api_key_env="ANTHROPIC_API_KEY", edge_mount="anthropic", edge_suffix="", @@ -146,6 +159,26 @@ class Cell: def registry_id(self, capability: Capability, streaming: Streaming, assertion: Assertion) -> str: return f"llm.{self.surface}.{self.deployment.route}.{capability}.{streaming}.{assertion}" + def subject(self, capability: Capability, streaming: Streaming, assertion: Assertion) -> Subject: + return Subject( + domain=Domain.SPEND_BUDGETS if assertion == "cost_logged" else Domain.LLM_TRANSLATION, + route=_surface_route(self.surface), + providers=(self.deployment.provider(),), + models=(self.deployment.backend,), + capabilities=() if capability == "basic" else (MetaCapability.FUNCTION_CALLING,), + mode=Mode.STREAM if streaming == "stream" else Mode.NONSTREAM, + ) + + +def _surface_route(surface: SurfaceName) -> Route: + match surface: + case "chat_completions": + return Route.CHAT_COMPLETIONS + case "messages": + return Route.MESSAGES + case "responses": + return Route.RESPONSES + CELLS: Final[tuple[Cell, ...]] = tuple( Cell(surface=surface, deployment=deployment, auth=auth) @@ -158,7 +191,14 @@ CELLS: Final[tuple[Cell, ...]] = tuple( def cells_covering(capability: Capability, streaming: Streaming, assertion: Assertion) -> tuple[ParameterSet, ...]: """Every cell as a pytest param carrying the registry id its test proves.""" return tuple( - pytest.param(cell, id=cell.id, marks=pytest.mark.covers(cell.registry_id(capability, streaming, assertion))) + pytest.param( + cell, + id=cell.id, + marks=( + pytest.mark.covers(cell.registry_id(capability, streaming, assertion)), + meta(cell.subject(capability, streaming, assertion)), + ), + ) for cell in CELLS ) @@ -352,9 +392,14 @@ class ChatCompletionsSurface: cost_header=response_header(raw.headers, "x-litellm-response-cost"), ) + @step( + 'Send a /chat/completions request to {model} with the prompt "{prompt}"' + " and forced weather tool use set to {with_tool}" + ) def reply(self, key: str, model: str, prompt: str, *, with_tool: bool = False) -> Reply: return self._turn(key, model, _chat_history(prompt), "forced" if with_tool else "none") + @step('Send a streaming /chat/completions request to {model} with the prompt "{prompt}"') def stream(self, key: str, model: str, prompt: str) -> StreamedReply: chunks: Final[tuple[ChatCompletionChunk, ...]] = tuple( self.sdk.openai(key).chat.completions.create( @@ -373,6 +418,7 @@ class ChatCompletionsSurface: event_count=len(chunks), ) + @step("Send the {call.name} tool result back to {model} over /chat/completions") def reply_to_tool_result(self, key: str, model: str, prompt: str, call: ToolCall, result: str) -> Reply: tool_call: Final[ChatCompletionMessageFunctionToolCallParam] = { "id": call.call_id, @@ -421,9 +467,14 @@ class MessagesSurface: cost_header=response_header(raw.headers, "x-litellm-response-cost"), ) + @step( + 'Send a /v1/messages request to {model} with the prompt "{prompt}"' + " and forced weather tool use set to {with_tool}" + ) def reply(self, key: str, model: str, prompt: str, *, with_tool: bool = False) -> Reply: return self._turn(key, model, ({"role": "user", "content": prompt},), "forced" if with_tool else "none") + @step('Send a streaming /v1/messages request to {model} with the prompt "{prompt}"') def stream(self, key: str, model: str, prompt: str) -> StreamedReply: events: Final[tuple[RawMessageStreamEvent, ...]] = tuple( self.sdk.anthropic(key).messages.create( @@ -446,6 +497,7 @@ class MessagesSurface: event_count=len(events), ) + @step("Send the {call.name} tool result back to {model} over /v1/messages") def reply_to_tool_result(self, key: str, model: str, prompt: str, call: ToolCall, result: str) -> Reply: tool_use: Final[ToolUseBlockParam] = { "type": "tool_use", @@ -496,9 +548,14 @@ class ResponsesSurface: cost_header=response_header(raw.headers, "x-litellm-response-cost"), ) + @step( + 'Send a /v1/responses request to {model} with the prompt "{prompt}"' + " and forced weather tool use set to {with_tool}" + ) def reply(self, key: str, model: str, prompt: str, *, with_tool: bool = False) -> Reply: return self._turn(key, model, [{"role": "user", "content": prompt}], "forced" if with_tool else "none") + @step('Send a streaming /v1/responses request to {model} with the prompt "{prompt}"') def stream(self, key: str, model: str, prompt: str) -> StreamedReply: events: Final[tuple[ResponseStreamEvent, ...]] = tuple( self.sdk.openai(key).responses.create( @@ -519,6 +576,7 @@ class ResponsesSurface: event_count=len(events), ) + @step("Send the {call.name} tool result back to {model} over /v1/responses") def reply_to_tool_result(self, key: str, model: str, prompt: str, call: ToolCall, result: str) -> Reply: function_call: Final[ResponseFunctionToolCallParam] = { "type": "function_call", diff --git a/tests/e2e/llm_translation/passthrough_client.py b/tests/e2e/llm_translation/passthrough_client.py index a56d3dc077e..478148e45aa 100644 --- a/tests/e2e/llm_translation/passthrough_client.py +++ b/tests/e2e/llm_translation/passthrough_client.py @@ -18,6 +18,7 @@ from websockets.exceptions import InvalidStatus from websockets.sync.client import connect from e2e_config import ws_base_url +from e2e_metadata import step from proxy_client import ProxyClient from e2e_http import FileUploadForm, Headers, NoBody, Result, StreamingResponse from models import ChatMessage @@ -34,7 +35,7 @@ class JsonSchema(BaseModel): class GeminiHeaders(Headers): - x_goog_api_key: str = Field(serialization_alias="x-goog-api-key") + x_goog_api_key: str = Field(serialization_alias="x-goog-api-key", repr=False) content_type: str = Field( default="application/json", serialization_alias="Content-Type" ) @@ -42,7 +43,7 @@ class GeminiHeaders(Headers): class AnthropicHeaders(Headers): - x_api_key: str = Field(serialization_alias="x-api-key") + x_api_key: str = Field(serialization_alias="x-api-key", repr=False) anthropic_version: str = Field( default="2023-06-01", serialization_alias="anthropic-version" ) @@ -56,7 +57,7 @@ class VertexHeaders(Headers): # Only the litellm virtual key; the /vertex_ai passthrough mints the Vertex token # from the proxy's own service account (the deployment marked use_in_pass_through), # so no upstream Authorization bearer is sent from the client. - x_litellm_api_key: str = Field(serialization_alias="x-litellm-api-key") + x_litellm_api_key: str = Field(serialization_alias="x-litellm-api-key", repr=False) content_type: str = Field( default="application/json", serialization_alias="Content-Type" ) @@ -246,6 +247,7 @@ class PassthroughClient: # ---- Gemini native passthrough (/gemini/v1beta/...) ----------------- + @step("Send a Gemini generateContent request to {model} through /gemini") def gemini_generate( self, key: str, @@ -263,6 +265,7 @@ class PassthroughClient: ), ) + @step("Send a Gemini streamGenerateContent request to {model} through /gemini") def gemini_stream( self, key: str, model: str, text: str, *, tags: list[str] | None = None ) -> StreamingResponse: @@ -278,6 +281,7 @@ class PassthroughClient: # ---- Vertex AI native passthrough (/vertex_ai/v1/projects/...) ------- + @step("Send a Vertex AI generateContent request to {model} in {location} through /vertex_ai") def vertex_generate( self, key: str, project: str, location: str, model: str, text: str ) -> StreamingResponse: @@ -295,6 +299,7 @@ class PassthroughClient: # ---- Anthropic native passthrough (/anthropic/v1/messages) ---------- + @step("Send a /v1/messages request to {model} through /anthropic with streaming set to {stream}") def anthropic_message( self, key: str, @@ -324,6 +329,7 @@ class PassthroughClient: # Relayed to OpenAI untouched, which is the whole point of the prefix: the # customer opts out of the gateway's managed-file handling here. + @step("Upload {filename} to /openai_passthrough/v1/files") def openai_passthrough_upload_file( self, key: str, *, content: bytes, filename: str ) -> Result[PassthroughFileObject]: @@ -336,6 +342,7 @@ class PassthroughClient: response_type=PassthroughFileObject, ) + @step("Delete the uploaded file through /openai_passthrough/v1/files") def openai_passthrough_delete_file( self, key: str, file_id: str ) -> Result[PassthroughFileDeleted]: @@ -346,6 +353,7 @@ class PassthroughClient: response_type=PassthroughFileDeleted, ) + @step("List batches from /openai_passthrough/v1/batches") def openai_passthrough_list_batches(self, key: str) -> Result[PassthroughBatchList]: return self.proxy.transport.get( "/openai_passthrough/v1/batches", @@ -360,6 +368,7 @@ class PassthroughClient: # budgets against this traffic, so a 200 that logs no spend is money the # gateway never sees. + @step("Send a /v1/responses request to {model} through /openai_passthrough with streaming set to {stream}") def openai_passthrough_responses( self, key: str, model: str, text: str, *, stream: bool = False ) -> StreamingResponse: @@ -370,6 +379,7 @@ class PassthroughClient: stream=stream, ) + @step('Send a /v1/embeddings request to {model} through /openai_passthrough for "{text}"') def openai_passthrough_embed( self, key: str, model: str, text: str ) -> StreamingResponse: @@ -379,6 +389,7 @@ class PassthroughClient: json=OpenAIEmbeddingBody(model=model, input=text), ) + @step("Send a /v1/chat/completions request to {model} through /openai") def openai_chat( self, key: str, model: str, text: str, *, max_completion_tokens: int = 64 ) -> StreamingResponse: @@ -397,6 +408,7 @@ class PassthroughClient: # The same prefixes over an upgrade instead of a POST, for the provider APIs # that only speak websocket (realtime, responses.connect). + @step("Open a websocket to {path} and wait for its first event") def openai_passthrough_websocket( self, key: str, diff --git a/tests/e2e/llm_translation/realtime/realtime_client.py b/tests/e2e/llm_translation/realtime/realtime_client.py index a280f4bc26b..6eae4fdf54d 100644 --- a/tests/e2e/llm_translation/realtime/realtime_client.py +++ b/tests/e2e/llm_translation/realtime/realtime_client.py @@ -307,9 +307,11 @@ def as_text(message: str | bytes) -> str: class RealtimeSession: connection: Connection - def send(self, event: BaseModel) -> None: + @step("Send the realtime event {event.type} over the websocket") + def send(self, event: SessionUpdate | ConversationItemCreate | ResponseCreate) -> None: self.connection.send(event.model_dump_json(by_alias=True, exclude_none=True)) + @step("Wait for a {stop_type} event on the realtime websocket") def collect_until( self, stop_type: str, *, timeout: float ) -> tuple[ReceivedEvent, ...]: @@ -381,6 +383,7 @@ class RealtimeSession: class RealtimeClient: proxy: ProxyClient + @step("Add a realtime deployment that calls {provider.litellm_params.model}") def provision(self, provider: RealtimeProvider) -> tuple[str, str]: """Register this provider's realtime deployment through /model/new and return (model_name, model_id). The name is marker-unique so it never collides with a @@ -393,6 +396,7 @@ class RealtimeClient: ) return model_name, model_id + @step("Open a /v1/realtime websocket session to {model}") @contextmanager def connect( self, *, key: str, model: str, timeout: float = 15.0 diff --git a/tests/e2e/llm_translation/realtime/test_realtime_bedrock_e2e.py b/tests/e2e/llm_translation/realtime/test_realtime_bedrock_e2e.py index 656882a4d92..a86f1daff63 100644 --- a/tests/e2e/llm_translation/realtime/test_realtime_bedrock_e2e.py +++ b/tests/e2e/llm_translation/realtime/test_realtime_bedrock_e2e.py @@ -19,6 +19,7 @@ from __future__ import annotations import pytest from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from realtime_client import ( @@ -42,6 +43,15 @@ class TestNovaSonicRealtime: "llm.realtime.bedrock_converse.basic.stream.works", exercised_on=["realtime"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider.BEDROCK,), + models=(NOVA_SONIC,), + mode=Mode.WEBSOCKET, + ) + ) def test_nova_sonic_response_create_completes( self, client: RealtimeClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/llm_translation/realtime/test_realtime_e2e.py b/tests/e2e/llm_translation/realtime/test_realtime_e2e.py index d7870b26497..622ed9d507f 100644 --- a/tests/e2e/llm_translation/realtime/test_realtime_e2e.py +++ b/tests/e2e/llm_translation/realtime/test_realtime_e2e.py @@ -12,7 +12,10 @@ hard failure, not a skip; once configured, a protocol failure is likewise a hard failure. See REALTIME_COVERAGE_MATRIX.md. """ +from typing import Final + import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from pydantic import BaseModel @@ -42,7 +45,42 @@ from websockets.exceptions import ConnectionClosedError pytestmark = pytest.mark.e2e -PROVIDER_PARAMS = [pytest.param(p, id=p.id) for p in PROVIDERS] +AZURE_REALTIME_MODEL: Final = "azure/gpt-realtime" + +TEXT_PARAMS: Final = tuple( + pytest.param( + p, + id=p.id, + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider(p.id),), + models=(p.litellm_params.model,), + mode=Mode.WEBSOCKET, + ) + ), + ) + for p in PROVIDERS +) + +TOOL_PARAMS: Final = tuple( + pytest.param( + p, + id=p.id, + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider(p.id),), + models=(p.litellm_params.model,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.WEBSOCKET, + ) + ), + ) + for p in PROVIDERS +) WEATHER_TOOL = FunctionTool( name="get_weather", @@ -62,7 +100,7 @@ class WeatherResult(BaseModel): temperature_f: int -@pytest.mark.parametrize("provider", PROVIDER_PARAMS) +@pytest.mark.parametrize("provider", TEXT_PARAMS) def test_text_conversation( client: RealtimeClient, scoped_key: str, @@ -99,7 +137,7 @@ def test_text_conversation( assert done.response.usage is not None, "response.done missing normalized usage" -@pytest.mark.parametrize("provider", PROVIDER_PARAMS) +@pytest.mark.parametrize("provider", TOOL_PARAMS) def test_tool_call_round_trip( client: RealtimeClient, scoped_key: str, @@ -158,7 +196,7 @@ _REFUSED_UPSTREAMS = ( "azure-bad-key", "azure-realtime-refused", LiteLLMParamsBody( - model="azure/gpt-realtime", + model=AZURE_REALTIME_MODEL, api_key="invalid-e2e-key", api_version="2025-08-28", realtime_protocol="GA", @@ -168,6 +206,15 @@ _REFUSED_UPSTREAMS = ( @pytest.mark.parametrize("provider", _REFUSED_UPSTREAMS, ids=[p.id for p in _REFUSED_UPSTREAMS]) +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider.AZURE,), + models=(AZURE_REALTIME_MODEL,), + mode=Mode.WEBSOCKET, + ) +) def test_upstream_handshake_refusal_is_an_error_event_and_policy_close( client: RealtimeClient, resources: ResourceManager, diff --git a/tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e.py b/tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e.py index 2e9cfcfe648..27d5abd1828 100644 --- a/tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e.py +++ b/tests/e2e/llm_translation/realtime/test_realtime_pipecat_audio_e2e.py @@ -24,10 +24,12 @@ Three test scenarios per provider: import asyncio import wave from pathlib import Path +from typing import Final import pytest from e2e_config import ws_base_url +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from realtime_client import ( PROVIDERS, RealtimeProvider, @@ -73,7 +75,59 @@ from pipecat.services.openai.realtime.llm import OpenAIRealtimeLLMService # noq from pipecat_service import LiteLLMRealtimeLLMService # noqa: E402 -PROVIDER_PARAMS = [pytest.param(p, id=p.id) for p in PROVIDERS] +TOOL_PARAMS: Final = tuple( + pytest.param( + p, + id=p.id, + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider(p.id),), + models=(p.litellm_params.model,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.WEBSOCKET, + ) + ), + ) + for p in PROVIDERS +) + +AUDIO_OUTPUT_PARAMS: Final = tuple( + pytest.param( + p, + id=p.id, + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider(p.id),), + models=(p.litellm_params.model,), + capabilities=(Capability.AUDIO_OUTPUT,), + mode=Mode.WEBSOCKET, + ) + ), + ) + for p in PROVIDERS +) + +AUDIO_INPUT_PARAMS: Final = tuple( + pytest.param( + p, + id=p.id, + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider(p.id),), + models=(p.litellm_params.model,), + capabilities=(Capability.AUDIO_INPUT, Capability.AUDIO_OUTPUT), + mode=Mode.WEBSOCKET, + ) + ), + ) + for p in PROVIDERS +) # PCM16 24 kHz mono WAV of "What is the weather in Paris?" (generated via macOS # `say` and resampled with audioop). Used by the server-VAD audio-input test. @@ -193,7 +247,7 @@ async def _run_pipeline( # --------------------------------------------------------------------------- -@pytest.mark.parametrize("provider", PROVIDER_PARAMS) +@pytest.mark.parametrize("provider", TOOL_PARAMS) def test_pipecat_server_vad( scoped_key: str, realtime_models: dict[str, str], @@ -208,7 +262,7 @@ def test_pipecat_server_vad( assert got_text, "no assistant text frames produced" -@pytest.mark.parametrize("provider", PROVIDER_PARAMS) +@pytest.mark.parametrize("provider", AUDIO_OUTPUT_PARAMS) def test_pipecat_audio_output( scoped_key: str, realtime_models: dict[str, str], @@ -332,7 +386,7 @@ async def _run_audio_input_pipeline( return bool(capture.texts), capture.audio_bytes -@pytest.mark.parametrize("provider", PROVIDER_PARAMS) +@pytest.mark.parametrize("provider", AUDIO_INPUT_PARAMS) def test_pipecat_server_vad_audio_input( scoped_key: str, realtime_models: dict[str, str], diff --git a/tests/e2e/llm_translation/realtime/test_realtime_pipecat_e2e.py b/tests/e2e/llm_translation/realtime/test_realtime_pipecat_e2e.py index f84ce197f88..3c22ae1a928 100644 --- a/tests/e2e/llm_translation/realtime/test_realtime_pipecat_e2e.py +++ b/tests/e2e/llm_translation/realtime/test_realtime_pipecat_e2e.py @@ -26,6 +26,7 @@ import asyncio import pytest from e2e_config import ws_base_url +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from realtime_client import ( PROVIDERS, RealtimeProvider, @@ -64,7 +65,21 @@ from pipecat_service import LiteLLMRealtimeLLMService # noqa: E402 # pipecat-ai/pipecat#2544); raw-ws tool_call_round_trip[vertex_ai] is the # source of truth for that provider. Keep openai/azure/gemini here. PROVIDER_PARAMS = [ - pytest.param(p, id=p.id) for p in PROVIDERS if p.id != "vertex_ai" + pytest.param( + p, + id=p.id, + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider(p.id),), + models=(p.litellm_params.model,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.WEBSOCKET, + ) + ), + ) + for p in PROVIDERS if p.id != "vertex_ai" ] WEATHER_TOOL = ToolsSchema( diff --git a/tests/e2e/llm_translation/test_audio_speech_e2e.py b/tests/e2e/llm_translation/test_audio_speech_e2e.py index 75a3de86d13..c4c3f751828 100644 --- a/tests/e2e/llm_translation/test_audio_speech_e2e.py +++ b/tests/e2e/llm_translation/test_audio_speech_e2e.py @@ -10,8 +10,11 @@ the SDK refuses to send a request missing its required fields. from __future__ import annotations +from typing import Final + import pytest from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from e2e_http import assert_client_error from lifecycle import ResourceManager from models import LiteLLMParamsBody @@ -21,6 +24,9 @@ from sdk_clients import SdkClients, response_header pytestmark = pytest.mark.e2e +OPENAI_TTS_MODEL: Final = "openai/gpt-4o-mini-tts" +AWS_POLLY_MODEL: Final = "aws_polly/generative" + class _OptionalSpeechBody(BaseModel): model: str | None = None @@ -32,7 +38,7 @@ def _register_tts(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, model = f"e2e-speech-{unique_marker()}" model_id = proxy.create_model( model, - LiteLLMParamsBody(model="openai/gpt-4o-mini-tts", api_key="os.environ/OPENAI_API_KEY"), + LiteLLMParamsBody(model=OPENAI_TTS_MODEL, api_key="os.environ/OPENAI_API_KEY"), ) resources.defer(lambda: proxy.delete_model(model_id)) return model, resources.key() @@ -40,6 +46,15 @@ def _register_tts(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, class TestAudioSpeech: @pytest.mark.covers("llm.audio_speech.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_TTS_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_audio_speech_returns_audio( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -56,6 +71,15 @@ class TestAudioSpeech: assert response.content, "/audio/speech returned an empty body" @pytest.mark.covers("llm.audio_speech.openai.basic.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_TTS_MODEL,), + mode=Mode.STREAM, + ) + ) def test_audio_speech_streams_audio_chunks( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -90,6 +114,15 @@ class TestAudioSpeech: @pytest.mark.skip(reason="stage red: product gap, /v1/audio/speech 500s on missing input instead of 400") @pytest.mark.covers("llm.audio_speech.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_TTS_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_missing_input_returns_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -103,6 +136,12 @@ class TestAudioSpeech: @pytest.mark.skip(reason="stage red: product gap, /v1/audio/speech 500s on missing model instead of 400") @pytest.mark.covers("llm.audio_speech.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + ) + ) def test_missing_model_returns_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -116,6 +155,15 @@ class TestAudioSpeech: @pytest.mark.skip(reason="stage red: product gap, /v1/audio/speech 500s on invalid voice instead of surfacing the provider 4xx") @pytest.mark.covers("llm.audio_speech.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_TTS_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_invalid_voice_returns_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -129,6 +177,15 @@ class TestAudioSpeech: @pytest.mark.skip(reason="stage red: product gap, /v1/audio/speech 500s on empty input instead of surfacing the provider 4xx") @pytest.mark.covers("llm.audio_speech.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_TTS_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_empty_input_returns_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -145,6 +202,15 @@ MP3_PREFIXES = (b"ID3", b"\xff\xfb", b"\xff\xf3", b"\xff\xf2") class TestAwsPollySpeech: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.AWS_POLLY,), + models=(AWS_POLLY_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_polly_generative_voice_returns_mp3( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -152,7 +218,7 @@ class TestAwsPollySpeech: model_id = proxy.create_model( model, LiteLLMParamsBody( - model="aws_polly/generative", + model=AWS_POLLY_MODEL, aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID", aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY", aws_region_name="os.environ/AWS_REGION", diff --git a/tests/e2e/llm_translation/test_audio_transcriptions_e2e.py b/tests/e2e/llm_translation/test_audio_transcriptions_e2e.py index 725e15a0209..cd1ee8fb03c 100644 --- a/tests/e2e/llm_translation/test_audio_transcriptions_e2e.py +++ b/tests/e2e/llm_translation/test_audio_transcriptions_e2e.py @@ -17,6 +17,7 @@ from typing import Final import pytest from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from e2e_http import UnknownApiError, unwrap from lifecycle import ResourceManager from models import LiteLLMParamsBody @@ -30,6 +31,8 @@ WEATHER_WAV = ( Path(__file__).resolve().parent / "realtime" / "fixtures" / "weather_question_24k.wav" ) +OPENAI_TRANSCRIBE_MODEL: Final = "openai/gpt-4o-mini-transcribe" +OPENAI_WHISPER_MODEL: Final = "openai/whisper-1" MISSING_MODEL_PHRASES: Final = ("model=none", "invalid model", "model is required") @@ -47,7 +50,7 @@ def _register(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, str] model_id = proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-4o-mini-transcribe", api_key="os.environ/OPENAI_API_KEY" + model=OPENAI_TRANSCRIBE_MODEL, api_key="os.environ/OPENAI_API_KEY" ), ) resources.defer(lambda: proxy.delete_model(model_id)) @@ -56,6 +59,15 @@ def _register(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, str] class TestAudioTranscriptions: @pytest.mark.covers("llm.audio_transcriptions.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_TRANSCRIBE_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_audio_transcriptions_returns_text( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -72,6 +84,15 @@ class TestAudioTranscriptions: ) @pytest.mark.covers("llm.audio_transcriptions.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_TRANSCRIBE_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_missing_file_returns_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -98,6 +119,12 @@ class TestAudioTranscriptions: pytest.fail(f"empty audio expected a file-specific 400, got {other!r}") @pytest.mark.covers("llm.audio_transcriptions.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + ) + ) def test_missing_model_returns_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -143,7 +170,7 @@ class TestWhisperTranscriptionFormats: self, proxy: ProxyClient, resources: ResourceManager, form: _WhisperForm, response_type: type[R] ) -> R: model_id = proxy.create_model( - form.model, LiteLLMParamsBody(model="openai/whisper-1", api_key="os.environ/OPENAI_API_KEY") + form.model, LiteLLMParamsBody(model=OPENAI_WHISPER_MODEL, api_key="os.environ/OPENAI_API_KEY") ) resources.defer(lambda: proxy.delete_model(model_id)) return unwrap( @@ -158,12 +185,30 @@ class TestWhisperTranscriptionFormats: ) ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_WHISPER_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_vtt_format_returns_webvtt_transcript(self, proxy: ProxyClient, resources: ResourceManager) -> None: form = _WhisperForm(model=f"e2e-whisper-vtt-{unique_marker()}", response_format="vtt") transcript = self._upload(proxy, resources, form, _TranscriptionResult) assert transcript.text.lstrip().startswith("WEBVTT"), f"vtt transcript is not WebVTT: {transcript.text[:200]!r}" assert "weather" in transcript.text.lower(), f"vtt transcript lost the spoken words: {transcript.text!r}" + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(OPENAI_WHISPER_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_verbose_json_returns_word_timestamps(self, proxy: ProxyClient, resources: ResourceManager) -> None: form = _WhisperForm( model=f"e2e-whisper-verbose-{unique_marker()}", diff --git a/tests/e2e/llm_translation/test_bedrock_native_e2e.py b/tests/e2e/llm_translation/test_bedrock_native_e2e.py index 19c1be7b6db..a9b9e45858d 100644 --- a/tests/e2e/llm_translation/test_bedrock_native_e2e.py +++ b/tests/e2e/llm_translation/test_bedrock_native_e2e.py @@ -8,6 +8,7 @@ from __future__ import annotations import pytest from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from e2e_http import ( assert_client_error, require_successful_call, @@ -100,6 +101,15 @@ def _default_invoke() -> InvokeBody: class TestBedrockNative: @pytest.mark.covers("llm.bedrock_native.bedrock_converse.basic.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_converse_returns_assistant(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -113,6 +123,15 @@ class TestBedrockNative: assert any(part.text.strip() for part in response.output.message.content) @pytest.mark.covers("llm.bedrock_native.bedrock_converse.basic.stream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_converse_stream_returns_chunks(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -126,6 +145,15 @@ class TestBedrockNative: assert result.chunks > 0, "converse-stream returned no events" @pytest.mark.covers("llm.bedrock_native.bedrock_invoke.basic.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invoke_returns_message(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -138,6 +166,15 @@ class TestBedrockNative: assert any(part.text.strip() for part in response.content) @pytest.mark.covers("llm.bedrock_native.bedrock_invoke.basic.stream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_invoke_stream_returns_chunks(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -151,6 +188,15 @@ class TestBedrockNative: assert result.chunks > 0, "invoke stream returned no events" @pytest.mark.covers("llm.bedrock_native.bedrock_converse.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_converse_missing_messages_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -161,6 +207,15 @@ class TestBedrockNative: assert_client_error(result, "converse missing messages") @pytest.mark.covers("llm.bedrock_native.bedrock_converse.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_converse_empty_messages_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -171,6 +226,14 @@ class TestBedrockNative: assert_client_error(result, "converse empty messages") @pytest.mark.covers("llm.bedrock_native.bedrock_converse.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + mode=Mode.NONSTREAM, + ) + ) def test_converse_invalid_model_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: _, key = _register(proxy, resources) result = proxy.transport.send( @@ -183,6 +246,15 @@ class TestBedrockNative: ) @pytest.mark.covers("llm.bedrock_native.bedrock_invoke.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invoke_missing_messages_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -193,6 +265,15 @@ class TestBedrockNative: assert_client_error(result, "invoke missing messages") @pytest.mark.covers("llm.bedrock_native.bedrock_invoke.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invoke_missing_max_tokens_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -206,6 +287,15 @@ class TestBedrockNative: assert_client_error(result, "invoke missing max_tokens") @pytest.mark.covers("llm.bedrock_native.bedrock_invoke.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invoke_invalid_temperature_returns_client_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_bedrock_provider_matrix_e2e.py b/tests/e2e/llm_translation/test_bedrock_provider_matrix_e2e.py index 21333d39849..c80609edcdd 100644 --- a/tests/e2e/llm_translation/test_bedrock_provider_matrix_e2e.py +++ b/tests/e2e/llm_translation/test_bedrock_provider_matrix_e2e.py @@ -18,6 +18,7 @@ import pytest from pydantic import BaseModel from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from e2e_http import StreamingResponse, unwrap from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody @@ -98,6 +99,15 @@ class TestBedrockResponseHeaders: "llm.chat_completions.bedrock_converse.response_headers.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(CONVERSE_REGIONAL_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_request_id_header_surfaces( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -117,6 +127,15 @@ class TestBedrockResponseHeaders: "llm.chat_completions.bedrock_converse.response_headers.stream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(CONVERSE_REGIONAL_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_bedrock_request_id_header_surfaces_on_stream( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -158,6 +177,15 @@ class TestBedrockBatchDeploymentServesChat: "llm.chat_completions.bedrock_converse.batch_deployment.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(CONVERSE_REGIONAL_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_batch_s3_keys_do_not_break_chat( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -179,6 +207,15 @@ class TestBedrockBatchDeploymentServesChat: class TestBedrockInvokeRegionalModelIds: @pytest.mark.covers("llm.chat_completions.bedrock_invoke.basic.nonstream.works", exercised_on=[]) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(INVOKE_REGIONAL_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invoke_regional_id_completes( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -190,6 +227,15 @@ class TestBedrockInvokeRegionalModelIds: _assert_completion(response) @pytest.mark.covers("llm.chat_completions.bedrock_invoke.basic.stream.works", exercised_on=[]) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(INVOKE_REGIONAL_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_invoke_regional_id_streams( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -205,6 +251,15 @@ class TestBedrockInvokeRegionalModelIds: class TestBedrockOpenAIFamilyDefaultRoute: @pytest.mark.covers("llm.chat_completions.bedrock_converse.basic.nonstream.works", exercised_on=[]) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(OPENAI_FAMILY_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_family_model_id_completes_with_max_tokens( self, client: PassthroughClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_bedrock_web_search_server_tool_e2e.py b/tests/e2e/llm_translation/test_bedrock_web_search_server_tool_e2e.py index b4253a82dd8..6133fa960ff 100644 --- a/tests/e2e/llm_translation/test_bedrock_web_search_server_tool_e2e.py +++ b/tests/e2e/llm_translation/test_bedrock_web_search_server_tool_e2e.py @@ -36,6 +36,7 @@ from __future__ import annotations import pytest from anthropic.types import WebSearchTool20250305Param from e2e_config import unique_marker +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -62,6 +63,16 @@ class TestBedrockWebSearchServerTool: "ephemeral stack ships the config in this module's docstring." ) @pytest.mark.covers("llm.messages.bedrock_invoke.web_search_server_tool.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=(BEDROCK_INVOKE_BACKEND,), + capabilities=(Capability.WEB_SEARCH,), + mode=Mode.NONSTREAM, + ) + ) def test_web_search_server_tool_is_served( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: diff --git a/tests/e2e/llm_translation/test_cache_control.py b/tests/e2e/llm_translation/test_cache_control.py index bec65144c9f..33c52efe37e 100644 --- a/tests/e2e/llm_translation/test_cache_control.py +++ b/tests/e2e/llm_translation/test_cache_control.py @@ -43,6 +43,7 @@ import pytest from pydantic import BaseModel from e2e_config import unique_marker +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from e2e_http import Result, UnknownApiError, unwrap from lifecycle import ResourceManager from models import CacheControl, ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody, RichMessage, TextBlock, Usage @@ -226,6 +227,16 @@ class TestCacheControl: "llm.chat_completions.bedrock_converse.prompt_cache_5m.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_MODEL,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_prompt_caching_reads_cache( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -242,6 +253,16 @@ class TestCacheControl: "llm.chat_completions.vertex.prompt_cache_5m.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_MODEL,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_vertex_prompt_caching_reads_cache( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -266,6 +287,16 @@ class TestCacheControl: "llm.chat_completions.anthropic.prompt_cache_5m.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_MODEL,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_anthropic_prompt_caching_reads_cache( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -282,6 +313,16 @@ class TestCacheControl: "llm.chat_completions.openai.prompt_cache_5m.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_MODEL,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_prompt_caching_reads_cache( self, client: PassthroughClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_cache_control_injection_tool_calls_e2e.py b/tests/e2e/llm_translation/test_cache_control_injection_tool_calls_e2e.py index 1481185a601..0d796a32ef3 100644 --- a/tests/e2e/llm_translation/test_cache_control_injection_tool_calls_e2e.py +++ b/tests/e2e/llm_translation/test_cache_control_injection_tool_calls_e2e.py @@ -5,6 +5,7 @@ from typing import Final, Literal, TypeAlias import pytest from e2e_config import unique_marker from e2e_http import Result, unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ( CacheControl, @@ -162,7 +163,36 @@ def _assert_normal_completion(response: ChatResponse, model_name: str) -> None: @pytest.mark.parametrize( "backend", - (pytest.param("azure_foundry", id="azure-foundry"), pytest.param("vertex", id="vertex")), + ( + pytest.param( + "azure_foundry", + id="azure-foundry", + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.AZURE_AI,), + models=(AZURE_MODEL,), + capabilities=(Capability.FUNCTION_CALLING, Capability.PROMPT_CACHING), + mode=Mode.NONSTREAM, + ) + ), + ), + pytest.param( + "vertex", + id="vertex", + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_MODEL,), + capabilities=(Capability.FUNCTION_CALLING, Capability.PROMPT_CACHING), + mode=Mode.NONSTREAM, + ) + ), + ), + ), ) @pytest.mark.provider_live @pytest.mark.covers("llm.chat_completions.azure_foundry.basic.nonstream.works") diff --git a/tests/e2e/llm_translation/test_chat_completions_contract_e2e.py b/tests/e2e/llm_translation/test_chat_completions_contract_e2e.py index 09b484eb120..45b9fa44e52 100644 --- a/tests/e2e/llm_translation/test_chat_completions_contract_e2e.py +++ b/tests/e2e/llm_translation/test_chat_completions_contract_e2e.py @@ -8,6 +8,7 @@ from __future__ import annotations import pytest from e2e_config import provider_edge_base, unique_marker from e2e_http import StreamingResponse, assert_client_error, require_successful_call, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody from proxy_client import ProxyClient @@ -62,6 +63,15 @@ def _chat_status(proxy: ProxyClient, key: str, body: BaseModel) -> StreamingResp class TestChatCompletionsContract: @pytest.mark.covers("llm.chat_completions.openai.multi_turn.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_multi_turn_history_is_honored(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_chat_model(proxy, resources) turn1 = unwrap( @@ -106,6 +116,15 @@ class TestChatCompletionsContract: assert "84" in second, f"turn2 must answer 84 from history, got: {second!r}" @pytest.mark.covers("llm.chat_completions.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_success_response_matches_chat_completion_contract( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -131,6 +150,12 @@ class TestChatCompletionsContract: assert (message.content or "").strip(), f"content must be non-empty: {result.body[:300]}" @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + ) + ) def test_missing_model_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: _, key = _register_chat_model(proxy, resources) result = _chat_status( @@ -145,12 +170,30 @@ class TestChatCompletionsContract: ) @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_missing_messages_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_chat_model(proxy, resources) result = _chat_status(proxy, key, ChatMissingMessagesBody(model=model)) assert_client_error(result, "missing messages") @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_empty_messages_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_chat_model(proxy, resources) result = _chat_status( @@ -161,6 +204,15 @@ class TestChatCompletionsContract: assert_client_error(result, "empty messages") @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invalid_role_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_chat_model(proxy, resources) result = _chat_status( @@ -175,6 +227,15 @@ class TestChatCompletionsContract: assert_client_error(result, "invalid role") @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invalid_temperatures_return_client_errors(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_chat_model(proxy, resources) for temperature in (-0.1, 2.1, 3.0, 100.0): @@ -191,6 +252,15 @@ class TestChatCompletionsContract: assert_client_error(result, f"temperature={temperature}") @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invalid_max_completion_tokens_return_client_errors( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -208,6 +278,15 @@ class TestChatCompletionsContract: assert_client_error(result, f"max_completion_tokens={max_completion_tokens}") @pytest.mark.covers("llm.chat_completions.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_temperature_boundaries_succeed(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_chat_model(proxy, resources) for temperature in (0.0, 2.0): diff --git a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py index 90ac18c16ac..1e0a8cbb412 100644 --- a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py +++ b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py @@ -26,6 +26,7 @@ from pydantic import BaseModel from e2e_config import unique_marker from e2e_http import StreamingResponse, unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ( ChatBody, @@ -53,16 +54,40 @@ OPENAI_BACKEND = "openai/gpt-5.6" ANTHROPIC_BACKEND = "anthropic/claude-haiku-4-5-20251001" BEDROCK_CONVERSE_BACKEND = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0" BEDROCK_NOVA_BACKEND: Final = "bedrock/us.amazon.nova-2-lite-v1:0" +VERTEX_MISTRAL_BACKEND: Final = "vertex_ai/mistral-small-2503" +VERTEX_GPT_OSS_BACKEND: Final = "vertex_ai/openai/gpt-oss-120b-maas" VERTEX_PARTNER_BACKENDS: Final = ( pytest.param( - "vertex_ai/mistral-small-2503", - marks=pytest.mark.skip( - reason="the e2e Vertex project has no access to mistral-small-2503 (404 publisher model not found)" + VERTEX_MISTRAL_BACKEND, + marks=( + pytest.mark.skip( + reason="the e2e Vertex project has no access to mistral-small-2503 (404 publisher model not found)" + ), + meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_MISTRAL_BACKEND,), + mode=Mode.STREAM, + ) + ), ), ), pytest.param( - "vertex_ai/openai/gpt-oss-120b-maas", - marks=pytest.mark.skip(reason="never served by the e2e Vertex project (60s read timeout, no headers)"), + VERTEX_GPT_OSS_BACKEND, + marks=( + pytest.mark.skip(reason="never served by the e2e Vertex project (60s read timeout, no headers)"), + meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_GPT_OSS_BACKEND,), + mode=Mode.STREAM, + ) + ), + ), ), ) PDF_DOCUMENT_URL: Final = ( @@ -216,19 +241,58 @@ _PERSON_SCHEMA: dict[str, object] = { }, } -CHAT_MODELS: tuple[tuple[str, str], ...] = ( - ("gpt-5.5", "openai"), - ("claude-haiku-4-5", "anthropic"), - ("gemini-2.5-flash", "gemini"), +OPENAI_CHAT_MODEL: Final = "gpt-5.5" +ANTHROPIC_CHAT_MODEL: Final = "claude-haiku-4-5" +GEMINI_FLASH_MODEL: Final = "gemini-2.5-flash" + +CHAT_MODELS: Final = ( + pytest.param( + OPENAI_CHAT_MODEL, + "openai", + id=f"{OPENAI_CHAT_MODEL}-openai", + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_CHAT_MODEL,), + mode=Mode.NONSTREAM, + ) + ), + ), + pytest.param( + ANTHROPIC_CHAT_MODEL, + "anthropic", + id=f"{ANTHROPIC_CHAT_MODEL}-anthropic", + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_CHAT_MODEL,), + mode=Mode.NONSTREAM, + ) + ), + ), + pytest.param( + GEMINI_FLASH_MODEL, + "gemini", + id=f"{GEMINI_FLASH_MODEL}-gemini", + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(GEMINI_FLASH_MODEL,), + mode=Mode.NONSTREAM, + ) + ), + ), ) class TestChatCompletionsRegression: - @pytest.mark.parametrize( - ("model", "route"), - CHAT_MODELS, - ids=[f"{model}-{route}" for model, route in CHAT_MODELS], - ) + @pytest.mark.parametrize(("model", "route"), CHAT_MODELS) @pytest.mark.covers( "llm.chat_completions.openai.basic.nonstream.works", "llm.chat_completions.anthropic.basic.nonstream.works", @@ -272,6 +336,15 @@ class TestCohereChat: "llm.chat_completions.cohere.basic.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.COHERE,), + models=(COHERE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_cohere_chat_returns_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -316,6 +389,15 @@ class TestGeminiChatCompletions: "llm.chat_completions.gemini.basic.nonstream.cost_logged", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(GEMINI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_gemini_chat_returns_content_and_logs_cost( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -377,6 +459,15 @@ class TestVertexChatCompletions: "llm.chat_completions.vertex.basic.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_vertex_chat_returns_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -406,6 +497,16 @@ class TestVertexChatCompletions: "llm.chat_completions.vertex.tool_use.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_vertex_chat_returns_tool_call( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -435,6 +536,16 @@ class TestVertexChatCompletions: "llm.chat_completions.vertex.vision.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_BACKEND,), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) + ) def test_vertex_chat_vision_describes_image( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -452,6 +563,15 @@ class TestVertexChatCompletions: "llm.chat_completions.vertex.basic.stream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_vertex_chat_streams_real_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -494,6 +614,15 @@ class TestAzureOpenAIChatCompletions: "llm.chat_completions.azure_openai.basic.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.AZURE,), + models=(AZURE_OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_azure_openai_chat_returns_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -523,6 +652,16 @@ class TestAzureOpenAIChatCompletions: "llm.chat_completions.azure_openai.tool_use.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.AZURE,), + models=(AZURE_OPENAI_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_azure_openai_chat_returns_tool_call( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -554,6 +693,15 @@ class TestAzureFoundryChatCompletions: "llm.chat_completions.azure_foundry.basic.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.AZURE_AI,), + models=(AZURE_FOUNDRY_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_azure_foundry_chat_returns_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -596,6 +744,14 @@ class TestHostedVllmChat: "llm.chat_completions.hosted_vllm.basic.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.HOSTED_VLLM,), + mode=Mode.NONSTREAM, + ) + ) def test_hosted_vllm_chat_returns_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -651,6 +807,15 @@ class TestOpenAIChatCompletions: "llm.chat_completions.openai.basic.stream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_openai_chat_streams_real_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -678,6 +843,15 @@ class TestOpenAIChatCompletions: "llm.chat_completions.openai.basic.nonstream.cost_logged", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_chat_logs_cost( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -711,6 +885,16 @@ class TestOpenAIChatCompletions: "llm.chat_completions.openai.tool_use.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_chat_returns_tool_call( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -742,6 +926,16 @@ class TestOpenAIChatCompletions: "llm.chat_completions.openai.structured_output.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_chat_structured_output_conforms_to_schema( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -775,6 +969,16 @@ class TestOpenAIChatCompletions: "llm.chat_completions.openai.thinking.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_chat_reasoning_reports_reasoning_tokens( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -825,6 +1029,16 @@ class TestOpenAIChatCompletions: "llm.chat_completions.openai.vision.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_VISION_BACKEND,), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_chat_vision_describes_image( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -842,6 +1056,16 @@ class TestOpenAIChatCompletions: "llm.chat_completions.openai.tool_use.stream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) + ) def test_openai_chat_streams_tool_call( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -889,6 +1113,15 @@ class TestBedrockConverseChatCompletions: "llm.chat_completions.bedrock_converse.basic.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_converse_chat_returns_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -913,6 +1146,15 @@ class TestBedrockConverseChatCompletions: "llm.chat_completions.bedrock_converse.basic.stream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_bedrock_converse_chat_streams_real_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -936,6 +1178,16 @@ class TestBedrockConverseChatCompletions: "llm.chat_completions.bedrock_converse.tool_use.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_converse_chat_returns_tool_call( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -962,6 +1214,16 @@ class TestBedrockConverseChatCompletions: "llm.chat_completions.bedrock_converse.thinking.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_converse_chat_returns_reasoning( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -992,6 +1254,16 @@ class TestBedrockConverseChatCompletions: "llm.chat_completions.bedrock_converse.vision.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_converse_chat_vision_describes_image( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -1001,6 +1273,16 @@ class TestBedrockConverseChatCompletions: response = unwrap(client.proxy.chat(key, ChatBody(model=model, messages=_vision_messages(), max_tokens=32))) _assert_describes_cat(response) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_NOVA_BACKEND,), + capabilities=(Capability.PDF_INPUT,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_converse_reads_a_pdf_sent_by_url( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -1107,6 +1389,16 @@ class TestAnthropicChatCompletions: "llm.chat_completions.anthropic.structured_output.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) + ) def test_anthropic_chat_structured_output_conforms_to_schema( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -1140,6 +1432,16 @@ class TestAnthropicChatCompletions: "llm.chat_completions.anthropic.thinking.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_anthropic_chat_returns_thinking_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -1178,6 +1480,16 @@ class TestAnthropicChatCompletions: "llm.chat_completions.anthropic.vision.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) + ) def test_anthropic_chat_vision_describes_image( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -1191,6 +1503,15 @@ class TestAnthropicChatCompletions: "llm.chat_completions.anthropic.basic.stream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_anthropic_chat_streams_real_content( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -1214,6 +1535,16 @@ class TestAnthropicChatCompletions: "llm.chat_completions.anthropic.tool_use.nonstream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_anthropic_chat_returns_tool_call( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -1240,6 +1571,16 @@ class TestAnthropicChatCompletions: "llm.chat_completions.anthropic.tool_use.stream.works", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) + ) def test_anthropic_chat_streams_tool_call( self, client: PassthroughClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_chat_mid_conversation_system_e2e.py b/tests/e2e/llm_translation/test_chat_mid_conversation_system_e2e.py index 480225b502e..71e716b2255 100644 --- a/tests/e2e/llm_translation/test_chat_mid_conversation_system_e2e.py +++ b/tests/e2e/llm_translation/test_chat_mid_conversation_system_e2e.py @@ -37,6 +37,7 @@ from pydantic import BaseModel from e2e_config import unique_marker from e2e_http import Result, unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import CacheControl, ChatResponse, LiteLLMParamsBody, RichMessage, TextBlock, Usage from passthrough_client import PassthroughClient @@ -281,6 +282,16 @@ class TestAnthropicChatMidConversationSystem: "llm.chat_completions.anthropic.mid_conversation_system.nonstream.cache_hit", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(FLAGGED_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_flagged_model_keeps_prompt_cache_across_system_reminder( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -290,6 +301,16 @@ class TestAnthropicChatMidConversationSystem: "llm.chat_completions.anthropic.mid_conversation_system.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(UNFLAGGED_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_unflagged_model_converts_system_reminder_and_succeeds( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -305,6 +326,16 @@ class TestBedrockInvokeChatMidConversationSystem: "llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.cache_hit", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(FLAGGED_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_flagged_model_keeps_prompt_cache_across_system_reminder( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -314,6 +345,16 @@ class TestBedrockInvokeChatMidConversationSystem: "llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(UNFLAGGED_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_unflagged_model_converts_system_reminder_and_succeeds( self, client: PassthroughClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_chat_stream_contract_e2e.py b/tests/e2e/llm_translation/test_chat_stream_contract_e2e.py index fdb76df703d..1e353d11283 100644 --- a/tests/e2e/llm_translation/test_chat_stream_contract_e2e.py +++ b/tests/e2e/llm_translation/test_chat_stream_contract_e2e.py @@ -5,6 +5,7 @@ from typing import Final import pytest from e2e_config import provider_edge_base, unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatStreamOptions, LiteLLMParamsBody, Usage from proxy_client import ProxyClient @@ -12,6 +13,8 @@ from pydantic import BaseModel pytestmark = [pytest.mark.e2e, pytest.mark.replayable] +OPENAI_BACKEND: Final = "openai/gpt-5.6" + class _Delta(BaseModel): content: str | None = None @@ -30,13 +33,22 @@ class _Chunk(BaseModel): class TestChatStreamContract: @pytest.mark.covers("llm.chat_completions.openai.basic.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_chat_stream_is_sse_and_ends_with_done(self, proxy: ProxyClient, resources: ResourceManager) -> None: model: Final = f"e2e-chat-stream-{unique_marker()}" base: Final = provider_edge_base("openai") model_id: Final = proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-5.6", + model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY", api_base=f"{base}/v1" if base else None, ), diff --git a/tests/e2e/llm_translation/test_chat_tool_round_trip_e2e.py b/tests/e2e/llm_translation/test_chat_tool_round_trip_e2e.py index 75ba5c23ff8..01d15aed443 100644 --- a/tests/e2e/llm_translation/test_chat_tool_round_trip_e2e.py +++ b/tests/e2e/llm_translation/test_chat_tool_round_trip_e2e.py @@ -6,6 +6,7 @@ from typing import Final import pytest from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ( ChatAssistantTurn, @@ -138,22 +139,72 @@ def _assert_tool_results_reach_the_model( class TestChatToolResultRoundTrip: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(GEMINI_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_gemini(self, client: PassthroughClient, resources: ResourceManager) -> None: model, key = _register(client, resources, _api_key_params(GEMINI_BACKEND, "GEMINI_API_KEY")) _assert_tool_results_reach_the_model(client, key, model, thinking=None, tool_choice="required") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.MISTRAL,), + models=(MISTRAL_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_mistral(self, client: PassthroughClient, resources: ResourceManager) -> None: model, key = _register(client, resources, _api_key_params(MISTRAL_BACKEND, "MISTRAL_API_KEY")) _assert_tool_results_reach_the_model(client, key, model, thinking=None, tool_choice="required") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_converse(self, client: PassthroughClient, resources: ResourceManager) -> None: model, key = _register(client, resources, _bedrock_params(BEDROCK_CONVERSE_BACKEND)) _assert_tool_results_reach_the_model(client, key, model, thinking=None, tool_choice="required") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING, Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_anthropic_with_extended_thinking(self, client: PassthroughClient, resources: ResourceManager) -> None: model, key = _register(client, resources, _api_key_params(ANTHROPIC_BACKEND, "ANTHROPIC_API_KEY")) _assert_tool_results_reach_the_model(client, key, model, thinking=THINKING, tool_choice=None) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_LEGACY_THINKING_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING, Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_converse_with_extended_thinking( self, client: PassthroughClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_completions_endpoint_e2e.py b/tests/e2e/llm_translation/test_completions_endpoint_e2e.py index 6202dada599..7f89d8a7ed8 100644 --- a/tests/e2e/llm_translation/test_completions_endpoint_e2e.py +++ b/tests/e2e/llm_translation/test_completions_endpoint_e2e.py @@ -10,8 +10,11 @@ the completion fails here. from __future__ import annotations +from typing import Final + import pytest from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -19,9 +22,20 @@ from sdk_clients import NO_PROXY_CACHE, SdkClients pytestmark = pytest.mark.e2e +OPENAI_COMPLETIONS_BACKEND: Final = "openai/gpt-5.4-nano" + class TestCompletionsEndpoint: @pytest.mark.covers("llm.completions.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_COMPLETIONS_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_text_completion_returns_text( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -29,7 +43,7 @@ class TestCompletionsEndpoint: model_id = proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-5.4-nano", + model=OPENAI_COMPLETIONS_BACKEND, api_key="os.environ/OPENAI_API_KEY", ), ) diff --git a/tests/e2e/llm_translation/test_containers_e2e.py b/tests/e2e/llm_translation/test_containers_e2e.py index 1c3e37ec8bb..0a494d11108 100644 --- a/tests/e2e/llm_translation/test_containers_e2e.py +++ b/tests/e2e/llm_translation/test_containers_e2e.py @@ -50,6 +50,7 @@ import openai import pytest from e2e_config import REQUEST_TIMEOUT, unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from management.management_client import ManagementClient, build_client from models import KeyGenerateBody, KeyGenerateResponse, LiteLLMParamsBody, TeamNewBody, UserNewBody @@ -172,6 +173,15 @@ def _assert_file_round_trip(client: OpenAI, native_id: str, marker: str) -> None class TestAzureContainerFiles: @pytest.mark.covers("llm.responses.azure_openai.code_interpreter.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CONTAINERS, + providers=(Provider.AZURE,), + models=(AZURE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_service_account_key_reads_container_file_by_native_id( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -187,6 +197,15 @@ class TestAzureContainerFiles: _assert_file_round_trip(client, native_id, marker) @pytest.mark.covers("llm.responses.azure_openai.code_interpreter.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CONTAINERS, + providers=(Provider.AZURE,), + models=(AZURE_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_service_account_key_reads_container_file_created_by_a_streamed_response( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -202,6 +221,13 @@ class TestAzureContainerFiles: class TestOpenAIContainerFiles: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CONTAINERS, + providers=(Provider.OPENAI,), + ) + ) def test_container_file_lifecycle_through_the_gateway(self, resources: ResourceManager, sdk: SdkClients) -> None: client: Final = sdk.openai(resources.key()) marker: Final = unique_marker() diff --git a/tests/e2e/llm_translation/test_credential_messages_e2e.py b/tests/e2e/llm_translation/test_credential_messages_e2e.py index 58e17f20bb8..55fd1cf7f2f 100644 --- a/tests/e2e/llm_translation/test_credential_messages_e2e.py +++ b/tests/e2e/llm_translation/test_credential_messages_e2e.py @@ -3,10 +3,12 @@ from __future__ import annotations import os +from typing import Final import pytest from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import CredentialCreateBody, LiteLLMParamsBody from proxy_client import ProxyClient @@ -14,9 +16,20 @@ from sdk_clients import NO_PROXY_CACHE, SdkClients pytestmark = pytest.mark.e2e +CLAUDE_BACKEND: Final = "anthropic/claude-haiku-4-5" + class TestCredentialBackedMessages: @pytest.mark.covers("mgmt.credential.new.serves_request") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(CLAUDE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_credential_backed_messages(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: marker = unique_marker() credential_name = f"e2e-cred-{marker}" @@ -35,7 +48,7 @@ class TestCredentialBackedMessages: model_id = proxy.create_model( model, LiteLLMParamsBody( - model="anthropic/claude-haiku-4-5", + model=CLAUDE_BACKEND, litellm_credential_name=credential_name, ), ) diff --git a/tests/e2e/llm_translation/test_custom_pricing_e2e.py b/tests/e2e/llm_translation/test_custom_pricing_e2e.py index 1cebf90fa21..6d8cd24648e 100644 --- a/tests/e2e/llm_translation/test_custom_pricing_e2e.py +++ b/tests/e2e/llm_translation/test_custom_pricing_e2e.py @@ -23,6 +23,7 @@ import pytest from pydantic import BaseModel, RootModel from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from proxy_client import ProxyClient from e2e_http import Success, unwrap from lifecycle import ResourceManager @@ -148,6 +149,15 @@ def _poll_breakdown_row(proxy: ProxyClient, key: str, response_id: str | None) - class TestCustomPricing: + @meta( + Subject( + domain=Domain.COST_MAP, + route=Route.SPEND_REPORTING, + providers=(Provider.GEMINI,), + models=(BACKEND_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_custom_pricing_is_billed_at_configured_rate( self, proxy: ProxyClient, @@ -193,6 +203,12 @@ class TestCustomPricing: f"= {completion * CUSTOM_OUTPUT_RATE}" ) + @meta( + Subject( + domain=Domain.COST_MAP, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_model_info_reports_custom_pricing( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -208,6 +224,12 @@ class TestCustomPricing: f"{entry.litellm_params.output_cost_per_token} != configured {CUSTOM_OUTPUT_RATE}" ) + @meta( + Subject( + domain=Domain.COST_MAP, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_custom_pricing_is_isolated_from_sibling_deployment( self, proxy: ProxyClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py b/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py index 8dfccf0d74b..8008345a3b6 100644 --- a/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py +++ b/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py @@ -20,6 +20,7 @@ from __future__ import annotations import pytest from e2e_config import unique_marker +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from e2e_http import unwrap from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody, ThinkingParam @@ -49,6 +50,16 @@ def _reasoning_content(response: ChatResponse) -> str | None: class TestDeepSeekReasoningDisable: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.DEEPSEEK,), + models=(REASONER,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_reasoner_returns_reasoning_by_default( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -71,6 +82,16 @@ class TestDeepSeekReasoningDisable: f"disable param, so the disable assertions below can't be trusted: {response}" ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.DEEPSEEK,), + models=(REASONER,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_reasoning_effort_none_disables_reasoning( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -93,6 +114,16 @@ class TestDeepSeekReasoningDisable: f"is still present: {response}" ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.DEEPSEEK,), + models=(REASONER,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_thinking_disabled_disables_reasoning( self, client: PassthroughClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_embeddings_endpoint_e2e.py b/tests/e2e/llm_translation/test_embeddings_endpoint_e2e.py index 0e5bac556cc..7b59ffbf15f 100644 --- a/tests/e2e/llm_translation/test_embeddings_endpoint_e2e.py +++ b/tests/e2e/llm_translation/test_embeddings_endpoint_e2e.py @@ -16,6 +16,7 @@ from typing import Final import pytest from e2e_config import provider_edge_base, unique_marker from e2e_http import assert_client_error +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -24,6 +25,10 @@ from sdk_clients import NO_PROXY_CACHE, SdkClients, response_header pytestmark = pytest.mark.e2e +OPENAI_EMBEDDING: Final = "openai/text-embedding-3-small" +BEDROCK_TITAN_EMBEDDING: Final = "bedrock/amazon.titan-embed-text-v2:0" +COHERE_EMBEDDING: Final = "cohere/embed-v4.0" +MISTRAL_EMBEDDING: Final = "mistral/mistral-embed" VERTEX_TEXT_EMBEDDING: Final = "vertex_ai/text-embedding-005" VERTEX_MULTIMODAL_EMBEDDING: Final = "vertex_ai/multimodalembedding@001" TOKENS_TEXT: Final = "The quick brown fox jumps over the lazy dog" @@ -45,7 +50,7 @@ def _cosine(left: list[float], right: list[float]) -> float: def _titan_params() -> LiteLLMParamsBody: return LiteLLMParamsBody( - model="bedrock/amazon.titan-embed-text-v2:0", + model=BEDROCK_TITAN_EMBEDDING, aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID", aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY", aws_region_name="os.environ/AWS_REGION", @@ -62,7 +67,7 @@ def _openai_embeddings_params() -> LiteLLMParamsBody: Vertex stay live: SigV4 signs the Host header, and neither has an edge mount.""" base = provider_edge_base("openai") return LiteLLMParamsBody( - model="openai/text-embedding-3-small", + model=OPENAI_EMBEDDING, api_key="os.environ/OPENAI_API_KEY", api_base=None if base is None else f"{base}/v1", ) @@ -97,10 +102,28 @@ def _assert_embedding_vector( class TestEmbeddingsEndpoint: @pytest.mark.replayable @pytest.mark.covers("llm.embeddings.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.OPENAI,), + models=(OPENAI_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_embeddings_returns_vector(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: _assert_embedding_vector(proxy, resources, sdk, "e2e-embeddings", _openai_embeddings_params()) @pytest.mark.covers("llm.embeddings.bedrock.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_TITAN_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_embeddings_returns_vector( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -113,6 +136,15 @@ class TestEmbeddingsEndpoint: ) @pytest.mark.covers("llm.embeddings.cohere.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.COHERE,), + models=(COHERE_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_cohere_embeddings_returns_vector( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -121,10 +153,19 @@ class TestEmbeddingsEndpoint: resources, sdk, "e2e-embeddings-cohere", - LiteLLMParamsBody(model="cohere/embed-v4.0", api_key="os.environ/COHERE_API_KEY"), + LiteLLMParamsBody(model=COHERE_EMBEDDING, api_key="os.environ/COHERE_API_KEY"), ) @pytest.mark.covers("llm.embeddings.vertex.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_TEXT_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_vertex_embeddings_returns_vector( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -134,12 +175,21 @@ class TestEmbeddingsEndpoint: sdk, "e2e-embeddings-vertex", LiteLLMParamsBody( - model="vertex_ai/text-embedding-005", + model=VERTEX_TEXT_EMBEDDING, vertex_project="os.environ/VERTEXAI_PROJECT", vertex_location="us-central1", ), ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.MISTRAL,), + models=(MISTRAL_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_mistral_embeddings_returns_vector( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -148,10 +198,19 @@ class TestEmbeddingsEndpoint: resources, sdk, "e2e-embeddings-mistral", - LiteLLMParamsBody(model="mistral/mistral-embed", api_key="os.environ/MISTRAL_API_KEY"), + LiteLLMParamsBody(model=MISTRAL_EMBEDDING, api_key="os.environ/MISTRAL_API_KEY"), ) @pytest.mark.covers("llm.embeddings.vertex.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_TEXT_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_vertex_embeddings_honor_requested_dimensions( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -166,6 +225,15 @@ class TestEmbeddingsEndpoint: assert len(embeddings.data[0].embedding) == 8, f"dimensions=8 was not honored: {embeddings!r}" assert embeddings.usage.prompt_tokens > 0, f"vertex embeddings reported no prompt usage: {embeddings.usage!r}" + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_MULTIMODAL_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_vertex_multimodal_embeddings_honor_dimensions_and_are_costed( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -181,6 +249,15 @@ class TestEmbeddingsEndpoint: cost = response_header(raw.headers, "x-litellm-response-cost") assert cost is not None and float(cost) > 0, f"multimodal embedding was not costed: {cost!r}" + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_TITAN_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_titan_embeds_token_array_input_as_its_decoded_text( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -199,6 +276,15 @@ class TestEmbeddingsEndpoint: @pytest.mark.replayable @pytest.mark.covers("llm.embeddings.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + providers=(Provider.OPENAI,), + models=(OPENAI_EMBEDDING,), + mode=Mode.NONSTREAM, + ) + ) def test_array_input_returns_vectors(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model, key = _register(proxy, resources, "e2e-embeddings-array", _openai_embeddings_params()) embeddings = sdk.openai(key).embeddings.create( @@ -208,6 +294,12 @@ class TestEmbeddingsEndpoint: @pytest.mark.replayable @pytest.mark.covers("llm.embeddings.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + ) + ) def test_missing_model_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() result = proxy.transport.send( @@ -219,6 +311,12 @@ class TestEmbeddingsEndpoint: @pytest.mark.replayable @pytest.mark.covers("llm.embeddings.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.EMBEDDINGS, + ) + ) def test_missing_input_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources, "e2e-embeddings-missin", _openai_embeddings_params()) result = proxy.transport.send( diff --git a/tests/e2e/llm_translation/test_files_batches_contract_e2e.py b/tests/e2e/llm_translation/test_files_batches_contract_e2e.py index 5627fa1c0bf..5e925f222b7 100644 --- a/tests/e2e/llm_translation/test_files_batches_contract_e2e.py +++ b/tests/e2e/llm_translation/test_files_batches_contract_e2e.py @@ -8,6 +8,7 @@ from __future__ import annotations import pytest from e2e_http import NoBody, Success, UnknownApiError, assert_client_error +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from proxy_client import ProxyClient from pydantic import BaseModel @@ -28,6 +29,13 @@ class BatchObject(BaseModel): class TestFilesBatchesContract: @pytest.mark.covers("llm.files.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + mode=Mode.BATCH, + ) + ) def test_upload_without_purpose_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() result = proxy.transport.upload( @@ -47,6 +55,13 @@ class TestFilesBatchesContract: pytest.fail(f"upload without purpose expected 4xx, got {other!r}") @pytest.mark.covers("llm.batches.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + mode=Mode.BATCH, + ) + ) def test_create_batch_missing_input_file_id_returns_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -59,6 +74,14 @@ class TestFilesBatchesContract: assert_client_error(result, "batch missing input_file_id") @pytest.mark.covers("llm.batches.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(Provider.OPENAI,), + mode=Mode.BATCH, + ) + ) def test_retrieve_invalid_batch_id_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() result = proxy.transport.get( diff --git a/tests/e2e/llm_translation/test_google_native_e2e.py b/tests/e2e/llm_translation/test_google_native_e2e.py index 6910519c6df..b18683bd4ad 100644 --- a/tests/e2e/llm_translation/test_google_native_e2e.py +++ b/tests/e2e/llm_translation/test_google_native_e2e.py @@ -13,6 +13,7 @@ from typing import Literal import pytest from e2e_config import unique_marker from e2e_http import StreamingResponse, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -85,6 +86,15 @@ def _streamed_text(result: StreamingResponse) -> str: class TestGoogleNativeGenerateContent: @pytest.mark.covers("llm.google_native.gemini.basic.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.GOOGLE_GENAI, + providers=(Provider.GEMINI,), + models=(UPSTREAM_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_generate_content_returns_response_cost_header( self, proxy: ProxyClient, @@ -104,6 +114,15 @@ class TestGoogleNativeGenerateContent: assert result.response_cost > 0, f"x-litellm-response-cost must be a real cost, got {result.response_cost}" @pytest.mark.covers("llm.google_native.gemini.basic.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.GOOGLE_GENAI, + providers=(Provider.GEMINI,), + models=(UPSTREAM_MODEL,), + mode=Mode.STREAM, + ) + ) def test_stream_generate_content_frames_sse_the_way_google_sdks_expect( self, proxy: ProxyClient, diff --git a/tests/e2e/llm_translation/test_image_edits_e2e.py b/tests/e2e/llm_translation/test_image_edits_e2e.py index e95b054862e..66b0ebfc538 100644 --- a/tests/e2e/llm_translation/test_image_edits_e2e.py +++ b/tests/e2e/llm_translation/test_image_edits_e2e.py @@ -11,10 +11,12 @@ as the `image` part, not a JSON body. The fixture image is a small generated from __future__ import annotations import base64 +from typing import Final import openai import pytest from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS, unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -22,6 +24,8 @@ from sdk_clients import SdkClients pytestmark = pytest.mark.e2e +IMAGE_EDIT_BACKEND: Final = "openai/gpt-image-1" + _TEST_PNG = base64.b64decode( "iVBORw0KGgoAAAANSUhEUgAAAEAAAABACAIAAAAlC+aJAAAAS0lEQVR42u3PMQ0AAAwDoPo3" "3UrYvQQckD4XAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEB" @@ -33,7 +37,7 @@ def _register_image_model(proxy: ProxyClient, resources: ResourceManager) -> tup model = f"e2e-image-edit-{unique_marker()}" model_id = proxy.create_model( model, - LiteLLMParamsBody(model="openai/gpt-image-1", api_key="os.environ/OPENAI_API_KEY"), + LiteLLMParamsBody(model=IMAGE_EDIT_BACKEND, api_key="os.environ/OPENAI_API_KEY"), ) resources.defer(lambda: proxy.delete_model(model_id)) return model, resources.key() @@ -49,6 +53,15 @@ def _assert_client_error(error: openai.APIStatusError, context: str) -> None: class TestImageEdit: @pytest.mark.covers("llm.images_edits.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(IMAGE_EDIT_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_image_edit_returns_image(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model, key = _register_image_model(proxy, resources) client = sdk.openai(key) @@ -64,6 +77,15 @@ class TestImageEdit: assert first.b64_json or first.url, f"edited image has neither b64_json nor url: {first!r}" @pytest.mark.covers("llm.images_edits.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(IMAGE_EDIT_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_empty_prompt_returns_error(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model, key = _register_image_model(proxy, resources) client = sdk.openai(key) @@ -73,6 +95,15 @@ class TestImageEdit: _assert_client_error(raised.value, "empty image-edit prompt") @pytest.mark.covers("llm.images_edits.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(IMAGE_EDIT_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_empty_image_returns_error(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model, key = _register_image_model(proxy, resources) client = sdk.openai(key) diff --git a/tests/e2e/llm_translation/test_image_generation_e2e.py b/tests/e2e/llm_translation/test_image_generation_e2e.py index 1db40e7e15a..15cf20aa65a 100644 --- a/tests/e2e/llm_translation/test_image_generation_e2e.py +++ b/tests/e2e/llm_translation/test_image_generation_e2e.py @@ -8,9 +8,12 @@ from litellm-regression-tests/tests/test_inference_endpoints.py. from __future__ import annotations +from typing import Final + import pytest from e2e_config import unique_marker from e2e_http import assert_client_error +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from openai.types import ImagesResponse @@ -20,6 +23,9 @@ from sdk_clients import SdkClients pytestmark = pytest.mark.e2e +OPENAI_IMAGE_BACKEND: Final = "openai/gpt-image-1-mini" +BEDROCK_IMAGE_BACKEND: Final = "bedrock/amazon.nova-canvas-v1:0" + class _OptionalImageBody(BaseModel): model: str | None = None @@ -47,12 +53,21 @@ def _register_openai_image(proxy: ProxyClient, resources: ResourceManager) -> tu proxy, resources, "e2e-image", - LiteLLMParamsBody(model="openai/gpt-image-1-mini", api_key="os.environ/OPENAI_API_KEY"), + LiteLLMParamsBody(model=OPENAI_IMAGE_BACKEND, api_key="os.environ/OPENAI_API_KEY"), ) class TestImageGeneration: @pytest.mark.covers("llm.images_generations.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(OPENAI_IMAGE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_image_generation_returns_image( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -61,6 +76,15 @@ class TestImageGeneration: _assert_image_returned(images) @pytest.mark.covers("llm.images_generations.bedrock.basic.nonstream.works", exercised_on=["images_generations"]) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + providers=(Provider.BEDROCK,), + models=(BEDROCK_IMAGE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_image_generation_returns_image( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -69,7 +93,7 @@ class TestImageGeneration: resources, "e2e-bedrock-image", LiteLLMParamsBody( - model="bedrock/amazon.nova-canvas-v1:0", + model=BEDROCK_IMAGE_BACKEND, aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID", aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY", aws_region_name="os.environ/AWS_REGION", @@ -80,6 +104,12 @@ class TestImageGeneration: @pytest.mark.skip(reason="stage red: product gap, /v1/images/generations 500s (aimage_generation TypeError) on missing prompt instead of 400") @pytest.mark.covers("llm.images_generations.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + ) + ) def test_missing_prompt_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_openai_image(proxy, resources) result = proxy.transport.send( @@ -90,6 +120,15 @@ class TestImageGeneration: assert_client_error(result, "images missing prompt") @pytest.mark.covers("llm.images_generations.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(OPENAI_IMAGE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_empty_prompt_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_openai_image(proxy, resources) result = proxy.transport.send( @@ -100,6 +139,15 @@ class TestImageGeneration: assert_client_error(result, "images empty prompt") @pytest.mark.covers("llm.images_generations.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(OPENAI_IMAGE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invalid_size_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_openai_image(proxy, resources) result = proxy.transport.send( @@ -110,6 +158,15 @@ class TestImageGeneration: assert_client_error(result, "images invalid size") @pytest.mark.covers("llm.images_generations.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(OPENAI_IMAGE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_invalid_n_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register_openai_image(proxy, resources) result = proxy.transport.send( diff --git a/tests/e2e/llm_translation/test_messages_azure_foundry_e2e.py b/tests/e2e/llm_translation/test_messages_azure_foundry_e2e.py index 7a99f9c45e1..b10b8fe8bea 100644 --- a/tests/e2e/llm_translation/test_messages_azure_foundry_e2e.py +++ b/tests/e2e/llm_translation/test_messages_azure_foundry_e2e.py @@ -14,6 +14,7 @@ import pytest from anthropic.types import RawMessageStreamEvent, ToolParam from e2e_config import unique_marker +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -56,6 +57,15 @@ class TestAzureFoundryMessages: return model @pytest.mark.covers("llm.messages.azure_foundry.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=(AZURE_FOUNDRY_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_basic_nonstream(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model = self._register(proxy, resources) client = sdk.anthropic(resources.key(models=[model])) @@ -71,6 +81,15 @@ class TestAzureFoundryMessages: assert text.strip(), f"/v1/messages returned no text: {message.content!r}" @pytest.mark.covers("llm.messages.azure_foundry.basic.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=(AZURE_FOUNDRY_MODEL,), + mode=Mode.STREAM, + ) + ) def test_basic_stream(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model = self._register(proxy, resources) client = sdk.anthropic(resources.key(models=[model])) @@ -85,6 +104,16 @@ class TestAzureFoundryMessages: _assert_streamed_ok([event.type for event in stream]) @pytest.mark.covers("llm.messages.azure_foundry.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=(AZURE_FOUNDRY_MODEL,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_tool_use_nonstream(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model = self._register(proxy, resources) client = sdk.anthropic(resources.key(models=[model])) @@ -102,6 +131,16 @@ class TestAzureFoundryMessages: ) @pytest.mark.covers("llm.messages.azure_foundry.tool_use.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=(AZURE_FOUNDRY_MODEL,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) + ) def test_tool_use_stream(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model = self._register(proxy, resources) client = sdk.anthropic(resources.key(models=[model])) @@ -122,6 +161,16 @@ class TestAzureFoundryMessages: ), "stream carried no tool_use block" assert "message_stop" in event_types, "stream never reached message_stop" + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=(AZURE_FOUNDRY_MODEL,), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) + ) def test_output_format_returns_schema_json( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: diff --git a/tests/e2e/llm_translation/test_messages_bedrock_e2e.py b/tests/e2e/llm_translation/test_messages_bedrock_e2e.py index 61a37af7fd0..4e37f01886a 100644 --- a/tests/e2e/llm_translation/test_messages_bedrock_e2e.py +++ b/tests/e2e/llm_translation/test_messages_bedrock_e2e.py @@ -5,6 +5,7 @@ from typing import Final import pytest from anthropic.types import RawContentBlockDeltaEvent, RawMessageDeltaEvent, TextBlock, TextDelta from e2e_config import unique_marker +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -33,6 +34,16 @@ def _register(proxy: ProxyClient, resources: ResourceManager, backend: str) -> s class TestBedrockMessages: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=(CONVERSE_CLAUDE_BACKEND,), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) + ) def test_converse_output_format_returns_schema_json_text( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -50,6 +61,15 @@ class TestBedrockMessages: assert_sentiment_json("".join(texts)) @pytest.mark.covers("llm.messages.bedrock_converse.basic.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=(NOVA_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_nova_stream_relays_text_usage_and_stop( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: diff --git a/tests/e2e/llm_translation/test_messages_e2e.py b/tests/e2e/llm_translation/test_messages_e2e.py index 90e74474d3b..6d2f690b947 100644 --- a/tests/e2e/llm_translation/test_messages_e2e.py +++ b/tests/e2e/llm_translation/test_messages_e2e.py @@ -17,6 +17,7 @@ from typing import Final import anthropic import pytest +from _pytest.mark.structures import ParameterSet from anthropic import Anthropic from anthropic.types import ( InputJSONDelta, @@ -42,6 +43,7 @@ from e2e_config import ( unique_marker, ) from e2e_http import assert_client_error +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import AnthropicErrorEvent, AnthropicMessagesBody, ChatMessage, LiteLLMParamsBody, SpendLogRow from provider_edge import EDGE_MOUNTS, LiveEdge, RunningEdge, StreamCut, start_provider_edge @@ -61,6 +63,7 @@ class _OptionalMessagesBody(BaseModel): ANTHROPIC_BACKEND = "anthropic/claude-haiku-4-5" +OPENAI_BRIDGE_BACKEND: Final = "openai/gpt-5.6" WEATHER_TOOL: ToolParam = { "name": "get_weather", @@ -110,6 +113,15 @@ def _user_turn(text: str) -> MessageParam: class TestAnthropicMessages: @pytest.mark.covers("llm.messages.anthropic.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_messages_returns_completion(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model, key = _register(proxy, resources) client = sdk.anthropic(key) @@ -121,6 +133,15 @@ class TestAnthropicMessages: assert _text(message).strip(), f"/v1/messages returned no text: {message.content!r}" @pytest.mark.covers("llm.messages.anthropic.basic.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_messages_logs_cost_matching_the_response_header( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -169,6 +190,15 @@ class TestAnthropicMessages: @pytest.mark.covers("llm.messages.anthropic.basic.stream.works") @pytest.mark.provider_live + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_messages_streams_completion(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: """Edge-wired like its non-streaming siblings, so record and replay both carry the streamed response. @@ -226,6 +256,16 @@ class TestAnthropicMessages: ) @pytest.mark.covers("llm.messages.anthropic.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_messages_tool_use(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model, key = _register(proxy, resources) client = sdk.anthropic(key) @@ -243,6 +283,16 @@ class TestAnthropicMessages: ) @pytest.mark.covers("llm.messages.anthropic.structured_output.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) + ) def test_messages_output_format_returns_schema_json( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -261,6 +311,14 @@ class TestAnthropicMessages: reason="stage red: product gap, /v1/messages 500s (anthropic_messages TypeError) on missing messages instead of 400" ) @pytest.mark.covers("llm.messages.anthropic.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(), + models=(), + ) + ) def test_missing_messages_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -274,6 +332,14 @@ class TestAnthropicMessages: reason="stage red: product gap, /v1/messages 500s (anthropic_messages TypeError) on missing max_tokens instead of 400" ) @pytest.mark.covers("llm.messages.anthropic.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(), + models=(), + ) + ) def test_missing_max_tokens_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) result = proxy.transport.send( @@ -284,6 +350,14 @@ class TestAnthropicMessages: assert_client_error(result, "messages missing max_tokens") @pytest.mark.covers("llm.messages.anthropic.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(), + models=(), + ) + ) def test_missing_model_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: _, key = _register(proxy, resources) result = proxy.transport.send( @@ -364,9 +438,26 @@ def _request_tool(client: Anthropic, model: str, question: MessageParam, tool: T return blocks[0] +def _openai_bridge_subject(mode: Mode) -> Subject: + return Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.OPENAI,), + models=(OPENAI_BRIDGE_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=mode, + ) + + class TestOpenAIMessagesToolContinuation: @pytest.mark.provider_live - @pytest.mark.parametrize("stream", [True, False], ids=["stream", "nonstream"]) + @pytest.mark.parametrize( + "stream", + [ + pytest.param(stream, marks=meta(_openai_bridge_subject(mode)), id=name) + for stream, name, mode in ((True, "stream", Mode.STREAM), (False, "nonstream", Mode.NONSTREAM)) + ], + ) def test_required_tool_arguments_and_correlated_result( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, stream: bool ) -> None: @@ -375,7 +466,7 @@ class TestOpenAIMessagesToolContinuation: model_id: Final = proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-5.6", api_key="os.environ/OPENAI_API_KEY", api_base=f"{base}/v1" if base else None + model=OPENAI_BRIDGE_BACKEND, api_key="os.environ/OPENAI_API_KEY", api_base=f"{base}/v1" if base else None ), ) resources.defer(lambda: proxy.delete_model(model_id)) @@ -478,6 +569,30 @@ _DROPPED_BEFORE_FIRST_BYTE: Final[tuple[tuple[str, _CutRegistration, StreamCut], ) +_CUT_SUBJECTS: Final[MappingProxyType[_CutRegistration, Subject]] = MappingProxyType( + { + _register_cut_bedrock: Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=(BEDROCK_BACKEND,), + mode=Mode.STREAM, + ), + _register_cut_anthropic: Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + mode=Mode.STREAM, + ), + } +) + + +def _cut_params(cases: tuple[tuple[str, _CutRegistration, StreamCut], ...]) -> list[ParameterSet]: + return [pytest.param(register, cut, id=name, marks=meta(_CUT_SUBJECTS[register])) for name, register, cut in cases] + + def _payload(frame: str) -> JsonValue | None: try: return _FRAME_PAYLOAD.validate_json(frame) @@ -495,7 +610,7 @@ def _bare_error_frame(frame: str) -> bool: class TestMessagesUpstreamStreamFailure: @pytest.mark.covers("llm.messages.anthropic.upstream_stream_failure.stream.error_event") @pytest.mark.parametrize( - ("register", "cut"), [case[1:] for case in _DROPPED_UPSTREAMS], ids=[case[0] for case in _DROPPED_UPSTREAMS] + ("register", "cut"), _cut_params(_DROPPED_UPSTREAMS) ) def test_interrupted_upstream_stream_raises_in_the_anthropic_sdk( self, @@ -533,7 +648,7 @@ class TestMessagesUpstreamStreamFailure: @pytest.mark.covers("llm.messages.anthropic.upstream_stream_failure.stream.error_event") @pytest.mark.parametrize( - ("register", "cut"), [case[1:] for case in _DROPPED_UPSTREAMS], ids=[case[0] for case in _DROPPED_UPSTREAMS] + ("register", "cut"), _cut_params(_DROPPED_UPSTREAMS) ) def test_interrupted_upstream_stream_is_an_anthropic_error_event( self, proxy: ProxyClient, resources: ResourceManager, register: _CutRegistration, cut: StreamCut @@ -585,11 +700,7 @@ class TestMessagesUpstreamStreamFailure: ) @pytest.mark.covers("llm.messages.anthropic.upstream_stream_failure.stream.error_status") - @pytest.mark.parametrize( - ("register", "cut"), - [case[1:] for case in _DROPPED_BEFORE_FIRST_BYTE], - ids=[case[0] for case in _DROPPED_BEFORE_FIRST_BYTE], - ) + @pytest.mark.parametrize(("register", "cut"), _cut_params(_DROPPED_BEFORE_FIRST_BYTE)) def test_upstream_that_hangs_up_before_the_first_byte_raises_with_its_status_in_the_anthropic_sdk( self, proxy: ProxyClient, @@ -622,11 +733,7 @@ class TestMessagesUpstreamStreamFailure: ) @pytest.mark.covers("llm.messages.anthropic.upstream_stream_failure.stream.error_status") - @pytest.mark.parametrize( - ("register", "cut"), - [case[1:] for case in _DROPPED_BEFORE_FIRST_BYTE], - ids=[case[0] for case in _DROPPED_BEFORE_FIRST_BYTE], - ) + @pytest.mark.parametrize(("register", "cut"), _cut_params(_DROPPED_BEFORE_FIRST_BYTE)) def test_upstream_that_hangs_up_before_the_first_byte_is_a_json_error_with_its_status( self, proxy: ProxyClient, resources: ResourceManager, register: _CutRegistration, cut: StreamCut ) -> None: diff --git a/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py b/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py index e9b4b394996..84e0674d959 100644 --- a/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py +++ b/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py @@ -34,6 +34,7 @@ import pytest from anthropic import Anthropic from anthropic.types import Message, MessageParam, TextBlockParam from e2e_config import unique_marker +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -208,6 +209,16 @@ class TestBedrockInvokeMidConversationSystem: "llm.messages.bedrock_invoke.mid_conversation_system.nonstream.cache_hit", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=(FLAGGED_INVOKE_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_flagged_model_keeps_prompt_cache_across_system_reminder( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -234,6 +245,16 @@ class TestBedrockInvokeMidConversationSystem: "llm.messages.bedrock_invoke.mid_conversation_system.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=(UNFLAGGED_INVOKE_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_unflagged_model_converts_system_reminder_and_succeeds( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: diff --git a/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py b/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py index 9f5ed8b05da..44701b827f0 100644 --- a/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py +++ b/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py @@ -41,6 +41,7 @@ import pytest from anthropic import Anthropic from anthropic.types import Message, MessageParam, TextBlockParam from e2e_config import unique_marker +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -290,6 +291,16 @@ class TestAzureFoundryMidConversationSystem: "llm.messages.azure_foundry.mid_conversation_system.nonstream.cache_hit", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=(FLAGGED_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_flagged_model_keeps_prompt_cache_across_system_reminder( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -299,6 +310,16 @@ class TestAzureFoundryMidConversationSystem: "llm.messages.azure_foundry.mid_conversation_system.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=(UNFLAGGED_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_unflagged_model_converts_system_reminder_and_succeeds( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -322,6 +343,16 @@ class TestVertexMidConversationSystem: "llm.messages.vertex.mid_conversation_system.nonstream.cache_hit", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=(FLAGGED_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_flagged_model_keeps_prompt_cache_across_system_reminder( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -333,6 +364,16 @@ class TestVertexMidConversationSystem: "llm.messages.vertex.mid_conversation_system.nonstream.works", exercised_on=[], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=(UNFLAGGED_MODEL,), + capabilities=(Capability.MID_CONVERSATION_SYSTEM, Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_unflagged_model_converts_system_reminder_and_succeeds( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: diff --git a/tests/e2e/llm_translation/test_moderations_e2e.py b/tests/e2e/llm_translation/test_moderations_e2e.py index e936f7b335a..4d2d5ea8b79 100644 --- a/tests/e2e/llm_translation/test_moderations_e2e.py +++ b/tests/e2e/llm_translation/test_moderations_e2e.py @@ -9,9 +9,12 @@ negative stays on the shared transport, since the SDK refuses to send it. from __future__ import annotations +from typing import Final + import pytest from e2e_config import unique_marker from e2e_http import assert_client_error +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from openai.types import Moderation @@ -21,6 +24,7 @@ from sdk_clients import SdkClients pytestmark = pytest.mark.e2e +OPENAI_MODERATION_BACKEND: Final = "openai/omni-moderation-latest" VIOLENT_TEXT = "I am going to find you and kill you, and I will hurt everyone you love." BENIGN_TEXT = "I enjoyed the sunny afternoon and a relaxing walk in the park today." @@ -35,7 +39,7 @@ def _register_moderation_model(proxy: ProxyClient, resources: ResourceManager) - model_id = proxy.create_model( model, LiteLLMParamsBody( - model="openai/omni-moderation-latest", api_key="os.environ/OPENAI_API_KEY" + model=OPENAI_MODERATION_BACKEND, api_key="os.environ/OPENAI_API_KEY" ), ) resources.defer(lambda: proxy.delete_model(model_id)) @@ -52,6 +56,15 @@ def _flagged_categories(item: Moderation) -> tuple[str, ...]: class TestModerations: @pytest.mark.covers("llm.moderations.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MODERATIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_MODERATION_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_moderations_flags_violent_content( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -64,6 +77,15 @@ class TestModerations: assert item.flagged, f"violent text was not flagged: {item!r}" assert _flagged_categories(item), f"flagged result reported no true category: {item!r}" + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MODERATIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_MODERATION_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_moderations_passes_benign_content( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -79,6 +101,14 @@ class TestModerations: @pytest.mark.skip(reason="stage red: product gap, /v1/moderations 500s (KeyError 'input') on missing input instead of 400") @pytest.mark.covers("llm.moderations.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MODERATIONS, + providers=(), + models=(), + ) + ) def test_missing_input_returns_error( self, proxy: ProxyClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_ocr_rust_e2e.py b/tests/e2e/llm_translation/test_ocr_rust_e2e.py index b54a6ea010b..cdc3fe87431 100644 --- a/tests/e2e/llm_translation/test_ocr_rust_e2e.py +++ b/tests/e2e/llm_translation/test_ocr_rust_e2e.py @@ -25,6 +25,7 @@ from typing import Final, Protocol import pytest from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from e2e_http import ( PROVIDER_RATE_LIMIT_ATTEMPTS, RateLimitedError, @@ -60,6 +61,13 @@ TEST_IMAGE_URL = ( ) +MISTRAL_OCR_MODEL: Final = "mistral/mistral-ocr-latest" +AZURE_AI_OCR_MODEL: Final = "azure_ai/mistral-document-ai-2512" +AZURE_DOC_INTELLIGENCE_MODEL: Final = "azure_ai/doc-intelligence/prebuilt-layout" +VERTEX_OCR_MODEL: Final = "vertex_ai/mistral-ocr-2505" +COHERE_OCR_MODEL: Final = "cohere/parse-v5.0" + + class OcrProvider(Protocol): """One OCR provider's deployment config: its model id plus the os.environ/* credential references the proxy resolves at call time. Each provider owns which @@ -70,7 +78,7 @@ class OcrProvider(Protocol): @dataclass(frozen=True, slots=True) class MistralOcr: - model: str = "mistral/mistral-ocr-latest" + model: str = MISTRAL_OCR_MODEL def litellm_params(self) -> LiteLLMParamsBody: return LiteLLMParamsBody(model=self.model, api_key="os.environ/MISTRAL_API_KEY") @@ -96,7 +104,7 @@ class AzureDocIntelligenceOcr: AZURE_DOCUMENT_INTELLIGENCE_API_KEY, which the OCR config resolves from the doc-intelligence model name when api_base/api_key are left unset.""" - model: str = "azure_ai/doc-intelligence/prebuilt-layout" + model: str = AZURE_DOC_INTELLIGENCE_MODEL def litellm_params(self) -> LiteLLMParamsBody: return LiteLLMParamsBody(model=self.model) @@ -120,7 +128,7 @@ class VertexOcr: @dataclass(frozen=True, slots=True) class CohereOcr: - model: str = "cohere/parse-v5.0" + model: str = COHERE_OCR_MODEL def litellm_params(self) -> LiteLLMParamsBody: return LiteLLMParamsBody(model=self.model, api_key="os.environ/COHERE_API_KEY") @@ -141,7 +149,7 @@ RUST_OCR_CASES: tuple[_OcrCase, ...] = ( ), _OcrCase( "azure-ai", - AzureAiOcr("azure_ai/mistral-document-ai-2512"), + AzureAiOcr(AZURE_AI_OCR_MODEL), OcrDocument(type="document_url", document_url=TEST_PDF_URL), ), _OcrCase( @@ -151,13 +159,11 @@ RUST_OCR_CASES: tuple[_OcrCase, ...] = ( ), _OcrCase( "vertex-mistral", - VertexOcr("vertex_ai/mistral-ocr-2505", "us-central1"), + VertexOcr(VERTEX_OCR_MODEL, "us-central1"), OcrDocument(type="document_url", document_url=TEST_PDF_URL), ), ) -_CASE_IDS = tuple(case.suffix for case in RUST_OCR_CASES) - PDF_TEXT: Final = "test pdf file" IMAGE_TEXT: Final = "litellm" PDF_DOCUMENT: Final = OcrDocument(type="document_url", document_url=TEST_PDF_URL) @@ -175,14 +181,37 @@ class _OcrContentCase: OCR_CONTENT_CASES: Final = ( _OcrContentCase("mistral-pdf", MistralOcr(), PDF_DOCUMENT, PDF_TEXT), _OcrContentCase("mistral-image", MistralOcr(), IMAGE_DOCUMENT, IMAGE_TEXT), - _OcrContentCase("azure-ai-image", AzureAiOcr("azure_ai/mistral-document-ai-2512"), IMAGE_DOCUMENT, IMAGE_TEXT), + _OcrContentCase("azure-ai-image", AzureAiOcr(AZURE_AI_OCR_MODEL), IMAGE_DOCUMENT, IMAGE_TEXT), _OcrContentCase( - "vertex-mistral-image", VertexOcr("vertex_ai/mistral-ocr-2505", "us-central1"), IMAGE_DOCUMENT, IMAGE_TEXT + "vertex-mistral-image", VertexOcr(VERTEX_OCR_MODEL, "us-central1"), IMAGE_DOCUMENT, IMAGE_TEXT ), _OcrContentCase("cohere-image", CohereOcr(), IMAGE_DOCUMENT, IMAGE_TEXT), ) +def _ocr_subject(provider: OcrProvider) -> Subject: + match provider: + case MistralOcr(): + vendor, model = Provider.MISTRAL, MISTRAL_OCR_MODEL + case AzureAiOcr(): + vendor, model = Provider.AZURE_AI, AZURE_AI_OCR_MODEL + case AzureDocIntelligenceOcr(): + vendor, model = Provider.AZURE_AI, AZURE_DOC_INTELLIGENCE_MODEL + case VertexOcr(): + vendor, model = Provider.VERTEX_AI, VERTEX_OCR_MODEL + case CohereOcr(): + vendor, model = Provider.COHERE, COHERE_OCR_MODEL + case _: + raise TypeError(f"no OCR subject for {provider!r}") + return Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.OCR, + providers=(vendor,), + models=(model,), + mode=Mode.NONSTREAM, + ) + + def _assert_ocr_document(response: OcrResponse) -> None: assert response.object == "ocr", f"expected object='ocr', got {response.object!r}" assert response.model, "response missing the resolved model name" @@ -202,7 +231,10 @@ def _assert_provider_rate_limit_relayed(model: str, outcome: RateLimitedError) - class TestRustOcrGateway: - @pytest.mark.parametrize("case", RUST_OCR_CASES, ids=_CASE_IDS) + @pytest.mark.parametrize( + "case", + [pytest.param(case, marks=meta(_ocr_subject(case.provider)), id=case.suffix) for case in RUST_OCR_CASES], + ) def test_rust_ocr_response(self, proxy: ProxyClient, resources: ResourceManager, case: _OcrCase) -> None: model = f"rust-ocr-{case.suffix}-{unique_marker()}" model_id = proxy.create_model(model, case.provider.litellm_params()) @@ -219,6 +251,14 @@ class TestRustOcrGateway: @pytest.mark.skip(reason="stage red: product gap, /v1/ocr 500s (aocr TypeError) on missing document instead of 400") @pytest.mark.covers("llm.ocr.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.OCR, + providers=(), + models=(), + ) + ) def test_missing_document_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model = f"rust-ocr-val-{unique_marker()}" model_id = proxy.create_model(model, MistralOcr().litellm_params()) @@ -233,7 +273,13 @@ class TestRustOcrGateway: class TestOcrDocumentContent: - @pytest.mark.parametrize("case", OCR_CONTENT_CASES, ids=tuple(case.suffix for case in OCR_CONTENT_CASES)) + @pytest.mark.parametrize( + "case", + [ + pytest.param(case, marks=meta(_ocr_subject(case.provider)), id=case.suffix) + for case in OCR_CONTENT_CASES + ], + ) def test_ocr_reads_the_document_and_bills_its_pages( self, proxy: ProxyClient, resources: ResourceManager, case: _OcrContentCase ) -> None: diff --git a/tests/e2e/llm_translation/test_passthrough_e2e.py b/tests/e2e/llm_translation/test_passthrough_e2e.py index 447fe7d30d9..52d1c10d829 100644 --- a/tests/e2e/llm_translation/test_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_passthrough_e2e.py @@ -12,10 +12,13 @@ A passthrough call returning non-2xx fails hard (never a skip); once it returns 2xx, a missing or zero-cost SpendLogs row fails too. """ +from typing import Final + import pytest from e2e_config import CHEAP_OPENAI_MODEL, unique_marker from e2e_http import require_successful_call, unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ChatResponse, KeyGenerateBody, SpendLogRow from passthrough_client import ( @@ -29,6 +32,8 @@ from passthrough_client import ( completed_responses_object, ) +GEMINI_MODEL: Final = "gemini-2.5-flash" +ANTHROPIC_PASSTHROUGH_MODEL: Final = "claude-haiku-4-5" EMBEDDING_MODEL = "text-embedding-3-small" REALTIME_MODEL = "gpt-realtime-2" @@ -57,12 +62,21 @@ def _fetch_cost_breakdown(client: PassthroughClient, request_id: str | None) -> # ---- Gemini passthrough ------------------------------------------------ +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_gemini_passthrough_nonstreaming_logs_cost( client: PassthroughClient, scoped_key: str ) -> None: tag = f"e2e-passthrough-{unique_marker()}" result = client.gemini_generate( - scoped_key, "gemini-2.5-flash", "Say hello in one word", tags=[tag, "gemini"] + scoped_key, GEMINI_MODEL, "Say hello in one word", tags=[tag, "gemini"] ) require_successful_call(result) @@ -73,6 +87,15 @@ def test_gemini_passthrough_nonstreaming_logs_cost( @pytest.mark.skip(reason="stage red: product gap, native passthrough returns no x-litellm-response-cost or x-ratelimit-* headers") +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route( client: PassthroughClient, scoped_key: str ) -> None: @@ -82,7 +105,7 @@ def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_rout today, which makes native traffic invisible to the same tooling. """ result = client.gemini_generate( - scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}" + scoped_key, GEMINI_MODEL, f"Say hello in one word. {unique_marker()}" ) require_successful_call(result) @@ -102,10 +125,19 @@ def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_rout ) +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.STREAM, + ) +) def test_gemini_passthrough_streaming_logs_cost( client: PassthroughClient, scoped_key: str ) -> None: - result = client.gemini_stream(scoped_key, "gemini-2.5-flash", "Count to five") + result = client.gemini_stream(scoped_key, GEMINI_MODEL, "Count to five") require_successful_call(result) assert result.chunks > 0, "streaming passthrough produced no events" @@ -113,12 +145,22 @@ def test_gemini_passthrough_streaming_logs_cost( assert row.custom_llm_provider == "gemini" +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_gemini_passthrough_tool_call_logs_cost( client: PassthroughClient, scoped_key: str ) -> None: result = client.gemini_generate( scoped_key, - "gemini-2.5-flash", + GEMINI_MODEL, "What is the weather in Paris? Use the get_weather tool.", tools=[ GeminiTool( @@ -146,10 +188,19 @@ def test_gemini_passthrough_tool_call_logs_cost( # ---- Anthropic passthrough --------------------------------------------- +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_PASSTHROUGH_MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_anthropic_passthrough_nonstreaming_logs_cost( client: PassthroughClient, scoped_key: str ) -> None: - result = client.anthropic_message(scoped_key, "claude-haiku-4-5", "Say hello") + result = client.anthropic_message(scoped_key, ANTHROPIC_PASSTHROUGH_MODEL, "Say hello") require_successful_call(result) row = _fetch_cost_breakdown(client, anthropic_message_id(result)) @@ -157,11 +208,20 @@ def test_anthropic_passthrough_nonstreaming_logs_cost( assert "claude" in (row.model or "") +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_PASSTHROUGH_MODEL,), + mode=Mode.STREAM, + ) +) def test_anthropic_passthrough_streaming_logs_cost( client: PassthroughClient, scoped_key: str ) -> None: result = client.anthropic_message( - scoped_key, "claude-haiku-4-5", "Count to five", stream=True + scoped_key, ANTHROPIC_PASSTHROUGH_MODEL, "Count to five", stream=True ) require_successful_call(result) assert result.chunks > 0, "streaming passthrough produced no events" @@ -170,12 +230,22 @@ def test_anthropic_passthrough_streaming_logs_cost( assert row.custom_llm_provider == "anthropic" +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_PASSTHROUGH_MODEL,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_anthropic_passthrough_tool_call_logs_cost( client: PassthroughClient, scoped_key: str ) -> None: result = client.anthropic_message( scoped_key, - "claude-haiku-4-5", + ANTHROPIC_PASSTHROUGH_MODEL, "What is the weather in Paris? Use the get_weather tool.", tools=[ AnthropicTool( @@ -205,13 +275,21 @@ class TestPassthroughModelAllowlist: """ @pytest.mark.covers("other.auth.passthrough.model_allowlist_enforced") + @meta( + Subject( + domain=Domain.PROXY_AUTH, + route=Route.PASSTHROUGH, + providers=(), + models=(), + ) + ) def test_passthrough_denies_model_outside_key_allowlist( self, client: PassthroughClient, resources: ResourceManager ) -> None: - key = client.proxy.generate_key(KeyGenerateBody(models=["gemini-2.5-flash"])) + key = client.proxy.generate_key(KeyGenerateBody(models=[GEMINI_MODEL])) resources.defer(lambda: client.proxy.delete_key(key)) - result = client.anthropic_message(key, "claude-haiku-4-5", f"say hi {unique_marker()}") + result = client.anthropic_message(key, ANTHROPIC_PASSTHROUGH_MODEL, f"say hi {unique_marker()}") assert result.status_code == 403, ( "a key restricted to gemini-2.5-flash must be denied a claude passthrough call, " f"got {result.status_code}: {result.body[:300]}" @@ -230,6 +308,14 @@ class TestOpenAIPassthroughPrefix: """ @pytest.mark.covers("llm.files.openai.passthrough.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.OPENAI,), + models=(), + ) + ) def test_passthrough_prefix_uploads_a_file_to_openai( self, client: PassthroughClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -252,6 +338,14 @@ class TestOpenAIPassthroughPrefix: assert uploaded.bytes == len(content) @pytest.mark.covers("llm.batches.openai.passthrough.nonstream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.OPENAI,), + models=(), + ) + ) def test_passthrough_prefix_lists_batches_from_openai( self, client: PassthroughClient, scoped_key: str ) -> None: @@ -274,6 +368,15 @@ class TestOpenAIPassthroughSpend: """ @pytest.mark.covers("llm.responses.openai.passthrough.stream.cost_logged") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.STREAM, + ) + ) def test_streamed_responses_call_logs_its_cost( self, client: PassthroughClient, scoped_key: str ) -> None: @@ -318,6 +421,15 @@ class TestOpenAIPassthroughSpend: ) @pytest.mark.covers("llm.embeddings.openai.passthrough.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.OPENAI,), + models=(EMBEDDING_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_embeddings_call_logs_its_cost( self, client: PassthroughClient, scoped_key: str ) -> None: @@ -353,6 +465,15 @@ class TestOpenAIProviderPrefixChat: """ @pytest.mark.covers("llm.chat_completions.openai.passthrough.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_prefix_chat_returns_completion_and_logs_its_cost( self, client: PassthroughClient, scoped_key: str ) -> None: @@ -393,6 +514,15 @@ class TestOpenAIPassthroughWebsocket: """ @pytest.mark.covers("llm.realtime.openai.passthrough.stream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.OPENAI,), + models=(REALTIME_MODEL,), + mode=Mode.WEBSOCKET, + ) + ) def test_realtime_upgrade_reaches_openai_through_the_passthrough_prefix( self, client: PassthroughClient, scoped_key: str ) -> None: @@ -414,6 +544,15 @@ class TestOpenAIPassthroughWebsocket: ) @pytest.mark.covers("llm.responses.openai.passthrough_websocket.stream.works") + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.OPENAI,), + models=(), + mode=Mode.WEBSOCKET, + ) + ) def test_responses_upgrade_is_accepted_on_the_openai_prefix( self, client: PassthroughClient, scoped_key: str ) -> None: diff --git a/tests/e2e/llm_translation/test_passthrough_headers_e2e.py b/tests/e2e/llm_translation/test_passthrough_headers_e2e.py index 26f98774c63..66d51d122da 100644 --- a/tests/e2e/llm_translation/test_passthrough_headers_e2e.py +++ b/tests/e2e/llm_translation/test_passthrough_headers_e2e.py @@ -20,6 +20,7 @@ from pydantic import BaseModel, Field from e2e_config import unique_marker from e2e_http import AuthHeaders, NoBody, require_successful_call, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import AnthropicMessagesResponse, ChatMessage, KeyGenerateBody from passthrough_client import PassthroughClient @@ -136,6 +137,15 @@ class TestPassthroughHeaders: "other.config.passthrough.headers_forwarded", exercised_on=[], ) + @meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_static_and_x_pass_headers_reach_upstream( self, client: PassthroughClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/llm_translation/test_provider_features_e2e.py b/tests/e2e/llm_translation/test_provider_features_e2e.py index 2ea1d28748d..9a3de0aba7d 100644 --- a/tests/e2e/llm_translation/test_provider_features_e2e.py +++ b/tests/e2e/llm_translation/test_provider_features_e2e.py @@ -16,10 +16,13 @@ Prompt caching lives in test_cache_control.py. from __future__ import annotations +from typing import Final + import pytest from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, LiteLLMParamsBody from passthrough_client import PassthroughClient @@ -27,12 +30,22 @@ from passthrough_client import PassthroughClient pytestmark = pytest.mark.e2e SERVICE_TIER = "priority" +OPENAI_BACKEND: Final = "openai/gpt-5.5" class TestServiceTier: @pytest.mark.covers( "llm.chat_completions.openai.service_tier.nonstream.works", exercised_on=[] ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_openai_service_tier_is_echoed( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -40,7 +53,7 @@ class TestServiceTier: model_id = client.proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY" + model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY" ), ) resources.defer(lambda: client.proxy.delete_model(model_id)) diff --git a/tests/e2e/llm_translation/test_realtime_http_e2e.py b/tests/e2e/llm_translation/test_realtime_http_e2e.py index 9579ae13bbc..ee3ce3893c2 100644 --- a/tests/e2e/llm_translation/test_realtime_http_e2e.py +++ b/tests/e2e/llm_translation/test_realtime_http_e2e.py @@ -9,6 +9,7 @@ from __future__ import annotations import pytest from e2e_config import unique_marker from e2e_http import NoBody, assert_auth_denied, unwrap +from e2e_metadata import Domain, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from proxy_client import ProxyClient @@ -59,6 +60,14 @@ def _register(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, str] class TestRealtimeHttp: @pytest.mark.covers("llm.realtime.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.REALTIME, + providers=(Provider.OPENAI,), + models=(REALTIME_BACKEND,), + ) + ) def test_create_client_secret(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, key = _register(proxy, resources) secret = unwrap( @@ -82,6 +91,7 @@ class TestRealtimeHttp: assert secret.session.type in (None, "realtime"), f"unexpected session type: {secret.session.type}" @pytest.mark.covers("other.auth.realtime.missing_header_denied") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.REALTIME)) def test_client_secret_missing_auth_is_denied(self, proxy: ProxyClient, resources: ResourceManager) -> None: model, _ = _register(proxy, resources) result = proxy.transport.send( @@ -92,6 +102,7 @@ class TestRealtimeHttp: assert_auth_denied(result, "realtime client_secrets missing auth") @pytest.mark.covers("other.auth.realtime.missing_header_denied") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.REALTIME)) def test_calls_without_auth_is_denied(self, proxy: ProxyClient) -> None: result = proxy.transport.send( "/v1/realtime/calls", diff --git a/tests/e2e/llm_translation/test_rerank_e2e.py b/tests/e2e/llm_translation/test_rerank_e2e.py index 87b8618e6fb..9fafff03004 100644 --- a/tests/e2e/llm_translation/test_rerank_e2e.py +++ b/tests/e2e/llm_translation/test_rerank_e2e.py @@ -9,9 +9,12 @@ litellm-regression-tests/tests/test_inference_endpoints.py. from __future__ import annotations +from typing import Final + import pytest from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody, RerankBody, RerankResponse from proxy_client import ProxyClient @@ -24,6 +27,8 @@ DOCUMENTS = [ "Washington, D.C. is the capital of the United States.", "Capital punishment has existed in the United States since before it was a country.", ] +COHERE_RERANK_BACKEND: Final = "cohere/rerank-v3.5" +BEDROCK_RERANK_BACKEND: Final = "bedrock/arn:aws:bedrock:us-east-1::foundation-model/cohere.rerank-v3-5:0" QUERY = "What is the capital of the United States?" @@ -43,11 +48,20 @@ def _rerank_top_3(proxy: ProxyClient, key: str, model: str) -> RerankResponse: class TestRerank: @pytest.mark.covers("llm.rerank.cohere.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RERANK, + providers=(Provider.COHERE,), + models=(COHERE_RERANK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_rerank_scores_top_n(self, proxy: ProxyClient, resources: ResourceManager) -> None: model = f"e2e-rerank-{unique_marker()}" model_id = proxy.create_model( model, - LiteLLMParamsBody(model="cohere/rerank-v3.5", api_key="os.environ/COHERE_API_KEY"), + LiteLLMParamsBody(model=COHERE_RERANK_BACKEND, api_key="os.environ/COHERE_API_KEY"), ) resources.defer(lambda: proxy.delete_model(model_id)) key = resources.key() @@ -55,6 +69,15 @@ class TestRerank: _assert_top_n_scored(_rerank_top_3(proxy, key, model)) @pytest.mark.covers("llm.rerank.bedrock.basic.nonstream.works", exercised_on=["rerank"]) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RERANK, + providers=(Provider.BEDROCK,), + models=(BEDROCK_RERANK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_rerank_scores_top_n( self, proxy: ProxyClient, resources: ResourceManager ) -> None: @@ -62,7 +85,7 @@ class TestRerank: model_id = proxy.create_model( model, LiteLLMParamsBody( - model="bedrock/arn:aws:bedrock:us-east-1::foundation-model/cohere.rerank-v3-5:0", + model=BEDROCK_RERANK_BACKEND, aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID", aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY", aws_region_name="os.environ/AWS_REGION", diff --git a/tests/e2e/llm_translation/test_responses_bridge_streaming_e2e.py b/tests/e2e/llm_translation/test_responses_bridge_streaming_e2e.py index 75817340876..7826dc388ea 100644 --- a/tests/e2e/llm_translation/test_responses_bridge_streaming_e2e.py +++ b/tests/e2e/llm_translation/test_responses_bridge_streaming_e2e.py @@ -23,13 +23,14 @@ from pydantic import BaseModel, Field from e2e_config import unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatTool, ChatToolFunction, LiteLLMParamsBody from passthrough_client import PassthroughClient pytestmark = pytest.mark.e2e -RESPONSES_ONLY_BACKEND = "openai/gpt-5.3-codex" +RESPONSES_ONLY_BACKEND: Final = "openai/gpt-5.3-codex" class _BridgeToolCallFunction(BaseModel): @@ -99,6 +100,15 @@ class TestResponsesBridgeChatCompletionsStreaming: "llm.chat_completions.openai.basic.stream.bridge_shares_chunk_id", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(RESPONSES_ONLY_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_bridged_stream_shares_one_chunk_id( self, client: PassthroughClient, resources: ResourceManager, bridged_model: str ) -> None: @@ -124,6 +134,15 @@ class TestResponsesBridgeChatCompletionsStreaming: "llm.chat_completions.openai.basic.stream.bridge_streams_sse", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(RESPONSES_ONLY_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_bridged_stream_delivers_content_finish_reason_and_done( self, client: PassthroughClient, resources: ResourceManager, bridged_model: str ) -> None: @@ -149,6 +168,16 @@ class TestResponsesBridgeChatCompletionsStreaming: "llm.chat_completions.openai.tool_use.stream.bridge_streams_tool_call", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(RESPONSES_ONLY_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) + ) def test_bridged_stream_reassembles_tool_call( self, client: PassthroughClient, resources: ResourceManager, bridged_model: str ) -> None: diff --git a/tests/e2e/llm_translation/test_responses_e2e.py b/tests/e2e/llm_translation/test_responses_e2e.py index cc6f98dad50..00a1d3611f9 100644 --- a/tests/e2e/llm_translation/test_responses_e2e.py +++ b/tests/e2e/llm_translation/test_responses_e2e.py @@ -21,6 +21,7 @@ import openai import pytest from e2e_config import PROVIDER_EDGE_ADVERTISE_HOST, PROVIDER_EDGE_BIND_HOST, unique_marker from e2e_http import assert_client_error +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, LiteLLMParamsBody from openai.types.responses import ( @@ -52,7 +53,10 @@ class _OptionalResponsesBody(BaseModel): max_output_tokens: int | None = None -BEDROCK_CONVERSE_BACKEND = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0" +OPENAI_MINI_BACKEND: Final = "openai/gpt-4o-mini" +OPENAI_VISION_BACKEND: Final = "openai/gpt-4o" +ANTHROPIC_BACKEND: Final = "anthropic/claude-haiku-4-5" +BEDROCK_CONVERSE_BACKEND: Final = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0" VERTEX_BACKEND: Final = "vertex_ai/gemini-2.5-flash" AZURE_OPENAI_BACKEND: Final = "azure/gpt-5.4-nano" AZURE_OPENAI_API_VERSION: Final = "v1" @@ -100,11 +104,11 @@ WEATHER_TOOL: FunctionToolParam = { def _openai_params() -> LiteLLMParamsBody: - return LiteLLMParamsBody(model="openai/gpt-4o-mini", api_key="os.environ/OPENAI_API_KEY") + return LiteLLMParamsBody(model=OPENAI_MINI_BACKEND, api_key="os.environ/OPENAI_API_KEY") def _anthropic_params() -> LiteLLMParamsBody: - return LiteLLMParamsBody(model="anthropic/claude-haiku-4-5", api_key="os.environ/ANTHROPIC_API_KEY") + return LiteLLMParamsBody(model=ANTHROPIC_BACKEND, api_key="os.environ/ANTHROPIC_API_KEY") def _bedrock_params() -> LiteLLMParamsBody: @@ -160,6 +164,15 @@ class WeatherArguments(BaseModel): class TestResponses: @pytest.mark.covers("llm.responses.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_MINI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_returns_completion( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -172,6 +185,15 @@ class TestResponses: assert response.output_text.strip(), f"/responses returned no output text: {response.output!r}" @pytest.mark.covers("llm.responses.openai.basic.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_MINI_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_responses_streaming_returns_completion( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -194,6 +216,15 @@ class TestResponses: ) @pytest.mark.covers("llm.responses.openai.basic.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_MINI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_logs_cost(self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients) -> None: model = _register(proxy, resources, _openai_params()) client = sdk.openai(resources.key()) @@ -219,6 +250,16 @@ class TestResponses: assert "gpt-4o-mini" in (row.model or ""), f"unexpected spend row model: {row.model}" @pytest.mark.covers("llm.responses.openai.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_MINI_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_returns_function_call( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -235,13 +276,23 @@ class TestResponses: _assert_weather_call(response) @pytest.mark.covers("llm.responses.openai.vision.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_VISION_BACKEND,), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_vision_describes_image( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: model = _register( proxy, resources, - LiteLLMParamsBody(model="openai/gpt-4o", api_key="os.environ/OPENAI_API_KEY"), + LiteLLMParamsBody(model=OPENAI_VISION_BACKEND, api_key="os.environ/OPENAI_API_KEY"), ) client = sdk.openai(resources.key()) @@ -264,6 +315,15 @@ class TestResponses: ) @pytest.mark.covers("llm.responses.anthropic.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_anthropic_returns_completion( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -276,6 +336,16 @@ class TestResponses: assert response.output_text.strip(), f"/responses returned no output text: {response.output!r}" @pytest.mark.covers("llm.responses.anthropic.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_anthropic_returns_function_call( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -292,6 +362,15 @@ class TestResponses: _assert_weather_call(response) @pytest.mark.covers("llm.responses.bedrock_converse.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_bedrock_returns_completion( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -304,6 +383,16 @@ class TestResponses: assert response.output_text.strip(), f"/responses over bedrock returned no output text: {response.output!r}" @pytest.mark.covers("llm.responses.bedrock_converse.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_bedrock_returns_function_call( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -320,6 +409,15 @@ class TestResponses: _assert_weather_call(response) @pytest.mark.covers("llm.responses.vertex.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_vertex_returns_completion( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -332,6 +430,16 @@ class TestResponses: assert response.output_text.strip(), f"/responses over vertex returned no output text: {response.output!r}" @pytest.mark.covers("llm.responses.vertex.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_vertex_returns_function_call( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -349,6 +457,15 @@ class TestResponses: _assert_weather_call(response) @pytest.mark.covers("llm.responses.azure_openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.AZURE,), + models=(AZURE_OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_azure_openai_returns_completion( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -363,6 +480,16 @@ class TestResponses: ) @pytest.mark.covers("llm.responses.azure_openai.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.AZURE,), + models=(AZURE_OPENAI_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_azure_openai_returns_function_call( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -380,7 +507,37 @@ class TestResponses: _assert_weather_call(response) @pytest.mark.provider_edge_host - @pytest.mark.parametrize("endpoint", ["/v1/responses", "/v1/chat/completions"]) + @pytest.mark.parametrize( + "endpoint", + [ + pytest.param( + "/v1/responses", + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ), + id="/v1/responses", + ), + pytest.param( + "/v1/chat/completions", + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.BEDROCK,), + models=(BEDROCK_CONVERSE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ), + id="/v1/chat/completions", + ), + ], + ) def test_bedrock_forwards_allowed_safety_identifier_as_additional_model_request_field( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, endpoint: str ) -> None: @@ -442,6 +599,14 @@ class TestResponses: reason="stage red: product gap, /v1/responses 500s (aresponses TypeError) on missing input instead of 400" ) @pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_MINI_BACKEND,), + ) + ) def test_missing_input_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model = _register(proxy, resources, _openai_params(), prefix="e2e-responses-val") key = resources.key() @@ -453,6 +618,12 @@ class TestResponses: assert_client_error(result, "responses missing input") @pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + ) + ) def test_missing_model_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() result = proxy.transport.send( @@ -463,6 +634,14 @@ class TestResponses: assert_client_error(result, "responses missing model") @pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_MINI_BACKEND,), + ) + ) def test_empty_input_returns_client_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: model = _register(proxy, resources, _openai_params(), prefix="e2e-responses-val") key = resources.key() @@ -509,6 +688,16 @@ class TodayReport(BaseModel): class TestResponsesOpenAIHostedFeatures: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(REASONING_BACKEND,), + capabilities=(Capability.FUNCTION_CALLING, Capability.REASONING, Capability.RESPONSE_SCHEMA), + mode=Mode.NONSTREAM, + ) + ) def test_reasoning_items_replay_into_structured_output_after_tool_call( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -561,6 +750,15 @@ class TestResponsesOpenAIHostedFeatures: assert TOOL_DATE in report.today, f"structured output ignored the tool result: {report!r}" @pytest.mark.provider_live + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(SHELL_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_shell_tool_stream_surfaces_shell_call_and_its_output( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: diff --git a/tests/e2e/llm_translation/test_responses_retrieve_e2e.py b/tests/e2e/llm_translation/test_responses_retrieve_e2e.py index 063d014d1f5..b4a9634fbef 100644 --- a/tests/e2e/llm_translation/test_responses_retrieve_e2e.py +++ b/tests/e2e/llm_translation/test_responses_retrieve_e2e.py @@ -12,6 +12,7 @@ import openai import pytest from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, unique_marker from e2e_http import NoBody, Success, UnknownApiError, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody from openai.types.responses import ( @@ -27,6 +28,7 @@ from sdk_clients import NO_PROXY_CACHE, SdkClients pytestmark = pytest.mark.e2e OPENAI_BACKEND: Final = "openai/gpt-5.5" +OPENAI_MINI_BACKEND: Final = "openai/gpt-4o-mini" LONG_TASK: Final = "Write a numbered list counting from 1 to 400, one number per line, with a short word after each." CANCELLABLE_STATUSES: Final = frozenset({"queued", "in_progress"}) @@ -69,11 +71,20 @@ class TestResponsesRetrieve: reason="stage red: product gap (LIT-5446), retrieve returns a different id than the stored response (non-idempotent response-id re-encryption)" ) @pytest.mark.covers("llm.responses.openai.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_MINI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_store_and_retrieve_by_id(self, proxy: ProxyClient, resources: ResourceManager) -> None: model = f"e2e-resp-store-{unique_marker()}" model_id = proxy.create_model( model, - LiteLLMParamsBody(model="openai/gpt-4o-mini", api_key="os.environ/OPENAI_API_KEY"), + LiteLLMParamsBody(model=OPENAI_MINI_BACKEND, api_key="os.environ/OPENAI_API_KEY"), ) resources.defer(lambda: proxy.delete_model(model_id)) key = resources.key() @@ -103,6 +114,12 @@ class TestResponsesRetrieve: reason="stage red: product gap (LIT-5447), retrieving an unknown response id returns 400 (model=None) instead of 404" ) @pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + ) + ) def test_invalid_response_id_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() get_result = proxy.transport.get( @@ -135,6 +152,15 @@ def _input_texts(item: object) -> tuple[str, ...]: @pytest.mark.provider_live class TestStoredResponseLifecycle: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_input_items_list_the_stored_prompt( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -150,6 +176,15 @@ class TestStoredResponseLifecycle: texts = tuple(text for item in items for text in _input_texts(item)) assert any(marker in text for text in texts), f"input_items did not list the stored prompt: {items!r}" + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_deleted_response_is_no_longer_retrievable( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -171,6 +206,15 @@ class TestStoredResponseLifecycle: @pytest.mark.provider_live class TestBackgroundResponseCancel: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_cancel_background_response( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -185,6 +229,15 @@ class TestBackgroundResponseCancel: cancelled = client.responses.cancel(created.id) assert cancelled.status == "cancelled", f"cancel did not stop the response: {cancelled.status}" + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_cancel_background_streaming_response_by_streamed_id( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: diff --git a/tests/e2e/llm_translation/test_sail_e2e.py b/tests/e2e/llm_translation/test_sail_e2e.py index 7267052e12c..77003ec4198 100644 --- a/tests/e2e/llm_translation/test_sail_e2e.py +++ b/tests/e2e/llm_translation/test_sail_e2e.py @@ -15,6 +15,7 @@ from typing import Final, Literal import pytest from e2e_config import SLOW_PROVIDER_TIMEOUT_SECONDS, unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody, SpendLogRow from openai import OpenAI @@ -116,6 +117,15 @@ def _assert_spend_row_matches(proxy: ProxyClient, key: str, header_cost: float) class TestSailChatCompletions: @pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.SAIL,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) @pytest.mark.parametrize( ("service_tier", "billed_tier"), [("balanced", "balanced"), ("auto", "base")] ) @@ -150,6 +160,15 @@ class TestSailChatCompletions: _assert_spend_row_matches(proxy, key, header_cost) @pytest.mark.covers("llm.chat_completions.sail.service_tier.nonstream.drops_unknown_tier_and_bills_asap") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.SAIL,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) @pytest.mark.parametrize("service_tier", ["bogus", 5]) def test_unknown_service_tier_is_dropped_and_billed_asap( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients, service_tier: str | int @@ -176,6 +195,15 @@ class TestSailChatCompletions: class TestSailResponses: @pytest.mark.covers("llm.responses.sail.service_tier.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.RESPONSES, + providers=(Provider.SAIL,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_caller_completion_window_bills_its_rates( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: @@ -204,6 +232,15 @@ class TestSailResponses: class TestSailMessages: @pytest.mark.covers("llm.messages.sail.basic.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.SAIL,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_plain_call_returns_a_message( self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients ) -> None: diff --git a/tests/e2e/llm_translation/test_together_ai_e2e.py b/tests/e2e/llm_translation/test_together_ai_e2e.py index 874c6d77d19..d1ec2764a70 100644 --- a/tests/e2e/llm_translation/test_together_ai_e2e.py +++ b/tests/e2e/llm_translation/test_together_ai_e2e.py @@ -26,6 +26,7 @@ from typing import Final import pytest from e2e_config import STREAM_MIN_LEAD_SECONDS, provider_paces_stream, unique_marker from e2e_http import StreamingResponse, require_successful_call, unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ( AnthropicAssistantTurn, @@ -324,6 +325,15 @@ def _weather_call(client: PassthroughClient, key: str, model: str) -> OutMessage class TestTogetherChatCompletions: @pytest.mark.covers("llm.chat_completions.together_ai.thinking.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_reasoning_surfaces_as_reasoning_content( self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str ) -> None: @@ -347,6 +357,15 @@ class TestTogetherChatCompletions: assert message.content and "43" in message.content, f"answer lost: {message}" @pytest.mark.covers("llm.chat_completions.together_ai.thinking.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + capabilities=(Capability.REASONING,), + mode=Mode.STREAM, + ) + ) def test_reasoning_streams_as_reasoning_content_deltas( self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str ) -> None: @@ -369,6 +388,15 @@ class TestTogetherChatCompletions: assert "43" in content, f"streamed answer lost: {content!r}" @pytest.mark.covers("llm.chat_completions.together_ai.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_tool_call_is_returned( self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str ) -> None: @@ -376,6 +404,15 @@ class TestTogetherChatCompletions: _ = _weather_call_ids(_weather_call(client, key, model)) @pytest.mark.covers("llm.chat_completions.together_ai.tool_use.stream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) + ) def test_tool_call_is_streamed( self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str ) -> None: @@ -407,6 +444,15 @@ class TestTogetherChatCompletions: assert "paris" in args.location.lower(), f"streamed tool arguments lost the location: {args}" @pytest.mark.covers("llm.chat_completions.together_ai.multi_turn.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_tool_result_round_trip( self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str ) -> None: @@ -443,6 +489,16 @@ class TestTogetherChatCompletions: ) @pytest.mark.covers("llm.chat_completions.together_ai.thinking.nonstream.template_kwargs_forwarded") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + models=(HYBRID_REASONING_BACKEND,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_template_kwargs_reach_together( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -476,6 +532,16 @@ class TestTogetherChatCompletions: assert treatment.content and "43" in treatment.content, f"answer lost: {treatment}" @pytest.mark.covers("llm.chat_completions.together_ai.thinking.nonstream.replayed_reasoning_forwarded") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + models=(REASONING_REPLAY_BACKEND,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_replayed_reasoning_content_reaches_together( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -496,6 +562,14 @@ class TestTogetherChatCompletions: ) @pytest.mark.covers("llm.chat_completions.together_ai.basic.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + mode=Mode.NONSTREAM, + ) + ) def test_cost_header_and_spend_row_match_the_registry_price( self, client: PassthroughClient, @@ -551,6 +625,16 @@ class TestTogetherChatCompletions: ) @pytest.mark.covers("llm.chat_completions.together_ai.thinking.nonstream.effort_none_disables") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + models=(HYBRID_REASONING_BACKEND,), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) + ) def test_reasoning_effort_none_reaches_together( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -584,6 +668,16 @@ class TestTogetherChatCompletions: assert treatment.content and "43" in treatment.content, f"answer lost: {treatment}" @pytest.mark.covers("llm.chat_completions.together_ai.structured_output.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + models=(HYBRID_REASONING_BACKEND,), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) + ) def test_response_format_json_schema_shapes_the_reply( self, client: PassthroughClient, resources: ResourceManager ) -> None: @@ -608,6 +702,15 @@ class TestTogetherChatCompletions: assert person.name, f"schema-shaped reply carries an empty name: {message.content!r}" @pytest.mark.covers("llm.chat_completions.together_ai.prompt_cache_5m.nonstream.cost_logged") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.TOGETHER_AI,), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_cache_read_tokens_bill_at_the_cache_read_rate( self, client: PassthroughClient, @@ -705,6 +808,15 @@ def _messages_weather_call( class TestTogetherMessages: @pytest.mark.covers("llm.messages.together_ai.tool_use.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.TOGETHER_AI,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_tool_use_block_is_returned( self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str ) -> None: @@ -712,6 +824,15 @@ class TestTogetherMessages: _messages_weather_call(client, key, model) @pytest.mark.covers("llm.messages.together_ai.multi_turn.nonstream.works") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.TOGETHER_AI,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_tool_result_round_trip( self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str ) -> None: @@ -744,6 +865,14 @@ class TestTogetherMessages: @pytest.mark.covers("llm.messages.together_ai.basic.stream.works") @pytest.mark.provider_live + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.TOGETHER_AI,), + mode=Mode.STREAM, + ) + ) def test_streams_text_deltas( self, client: PassthroughClient, resources: ResourceManager, reasoning_tool_backend: str ) -> None: diff --git a/tests/e2e/llm_translation/test_token_counter_gemini_contents_e2e.py b/tests/e2e/llm_translation/test_token_counter_gemini_contents_e2e.py index b2b8f46ab0a..d50a7bae9d1 100644 --- a/tests/e2e/llm_translation/test_token_counter_gemini_contents_e2e.py +++ b/tests/e2e/llm_translation/test_token_counter_gemini_contents_e2e.py @@ -9,15 +9,43 @@ route the claude_code rows never reach from __future__ import annotations +from typing import Final + import pytest from e2e_config import unique_marker from e2e_http import require_successful_call +from e2e_metadata import Domain, Provider, Route, Subject, meta from proxy_client import ProxyClient from pydantic import BaseModel pytestmark = pytest.mark.e2e -GEMINI_DEPLOYMENTS = ("gemini-2.5-flash", "gemini-2.5-flash-vertex") +GEMINI_STUDIO_DEPLOYMENT: Final = "gemini-2.5-flash" +GEMINI_VERTEX_DEPLOYMENT: Final = "gemini-2.5-flash-vertex" +GEMINI_DEPLOYMENTS = ( + pytest.param( + GEMINI_STUDIO_DEPLOYMENT, + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.COUNT_TOKENS, + providers=(Provider.GEMINI,), + models=(GEMINI_STUDIO_DEPLOYMENT,), + ) + ), + ), + pytest.param( + GEMINI_VERTEX_DEPLOYMENT, + marks=meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.COUNT_TOKENS, + providers=(Provider.VERTEX_AI,), + models=(GEMINI_VERTEX_DEPLOYMENT,), + ) + ), + ), +) class _Part(BaseModel): diff --git a/tests/e2e/llm_translation/test_vector_stores_e2e.py b/tests/e2e/llm_translation/test_vector_stores_e2e.py index 71015d28d9f..2ae56594335 100644 --- a/tests/e2e/llm_translation/test_vector_stores_e2e.py +++ b/tests/e2e/llm_translation/test_vector_stores_e2e.py @@ -12,6 +12,7 @@ from typing import Literal import pytest from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, unique_marker +from e2e_metadata import Domain, Provider, Route, Subject, meta from e2e_http import ( FileUploadForm, NoBody, @@ -162,6 +163,7 @@ def _await_store_in_list(proxy: ProxyClient, key: str, store_id: str) -> None: class TestVectorStores: @pytest.mark.covers("llm.vector_stores.openai.basic.nonstream.works") + @meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,))) def test_create_list_retrieve_delete_lifecycle(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() name = f"e2e-vector-store-{unique_marker()}" @@ -204,6 +206,7 @@ class TestVectorStores: reason="stage red: product gap, vector store search 500s (asearch TypeError) on missing query instead of 400" ) @pytest.mark.covers("llm.vector_stores.openai.input_validation.nonstream.works") + @meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,))) def test_search_missing_query_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() created = unwrap( @@ -223,6 +226,7 @@ class TestVectorStores: assert_client_error(result, "vector store search missing query") @pytest.mark.covers("llm.vector_stores.openai.basic.nonstream.works") + @meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,))) def test_file_attach_poll_and_search(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() marker = f"azure-falcon-{unique_marker()}" @@ -309,6 +313,7 @@ class TestVectorStores: reason="stage red: product gap, retrieving a nonexistent vector store returns 2xx with an error envelope in the body instead of 404" ) @pytest.mark.covers("llm.vector_stores.openai.input_validation.nonstream.works") + @meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,))) def test_retrieve_invalid_id_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() result = proxy.transport.get( @@ -328,6 +333,7 @@ class TestVectorStores: pytest.fail(f"invalid vector store id must be a client error, got {other!r}") @pytest.mark.covers("llm.vector_stores.openai.input_validation.nonstream.works") + @meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.VECTOR_STORES, providers=(Provider.OPENAI,))) def test_invalid_chunking_returns_error(self, proxy: ProxyClient, resources: ResourceManager) -> None: key = resources.key() result = proxy.transport.send( diff --git a/tests/e2e/llm_translation/test_vertex_passthrough_e2e.py b/tests/e2e/llm_translation/test_vertex_passthrough_e2e.py index 5e9c9f614e5..3daf78a31b2 100644 --- a/tests/e2e/llm_translation/test_vertex_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_vertex_passthrough_e2e.py @@ -30,6 +30,7 @@ from pydantic import BaseModel from e2e_config import settle_propagation, unique_marker from e2e_http import NoBody, require_successful_call, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import SpendLogRow from passthrough_client import PassthroughClient @@ -149,6 +150,15 @@ def _costed_row(client: PassthroughClient, call_id: str | None) -> SpendLogRow: class TestVertexPassthroughSpendTracking: + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.PASSTHROUGH, + providers=(Provider.VERTEX_AI,), + models=(VERTEX_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_vertex_passthrough_via_managed_model_logs_cost( self, client: PassthroughClient, From 0ae55bdf7ad0bc81a73399f5e8e820ad2a109917 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Wed, 7 Oct 2026 10:31:54 -0700 Subject: [PATCH 03/13] test(e2e): tag claude_code cells with Subject metadata and record CLI driver steps (#44961) * test(e2e): add enum values, auto-discovering label gates and secret hiding for e2e metadata * test(e2e): tag claude_code tests with Subject metadata and record CLI driver steps * docs(e2e): name every markerless harness test file that carries no Subject * test(e2e): keep the step discovery comprehensions to one for clause * test(e2e): decorate run_claude directly so the label gate discovers its step --- tests/e2e/claude_code/_basic_messaging.py | 3 +++ tests/e2e/claude_code/_passthrough.py | 6 +++++ .../test_anthropic.py | 10 +++++++++ .../test_azure.py | 10 +++++++++ .../test_azure_openai.py | 10 +++++++++ .../test_bedrock_converse.py | 10 +++++++++ .../test_bedrock_invoke.py | 10 +++++++++ .../test_bedrock_mantle.py | 10 +++++++++ .../test_openai.py | 10 +++++++++ .../test_vertex_ai.py | 10 +++++++++ .../test_vertex_ai_gpt.py | 2 ++ .../test_anthropic.py | 10 +++++++++ .../basic_messaging_streaming/test_azure.py | 10 +++++++++ .../test_azure_openai.py | 10 +++++++++ .../test_bedrock_converse.py | 10 +++++++++ .../test_bedrock_invoke.py | 10 +++++++++ .../test_bedrock_mantle.py | 10 +++++++++ .../basic_messaging_streaming/test_openai.py | 10 +++++++++ .../test_vertex_ai.py | 10 +++++++++ .../test_vertex_ai_gpt.py | 2 ++ tests/e2e/claude_code/cli_driver.py | 4 ++++ .../count_tokens/test_anthropic.py | 10 +++++++++ .../claude_code/count_tokens/test_azure.py | 10 +++++++++ .../count_tokens/test_bedrock_converse.py | 10 +++++++++ .../count_tokens/test_bedrock_invoke.py | 10 +++++++++ .../count_tokens/test_vertex_ai.py | 10 +++++++++ tests/e2e/claude_code/http_probe.py | 7 ++++++ .../long_context_1m/test_anthropic.py | 10 +++++++++ .../claude_code/long_context_1m/test_azure.py | 10 +++++++++ .../long_context_1m/test_bedrock_converse.py | 10 +++++++++ .../long_context_1m/test_bedrock_invoke.py | 10 +++++++++ .../long_context_1m/test_vertex_ai.py | 10 +++++++++ .../claude_code/passthrough/test_anthropic.py | 10 +++++++++ .../e2e/claude_code/passthrough/test_azure.py | 10 +++++++++ .../passthrough/test_bedrock_converse.py | 3 +++ .../passthrough/test_bedrock_invoke.py | 10 +++++++++ .../claude_code/passthrough/test_vertex_ai.py | 10 +++++++++ .../claude_code/pdf_input/test_anthropic.py | 11 ++++++++++ tests/e2e/claude_code/pdf_input/test_azure.py | 11 ++++++++++ .../pdf_input/test_bedrock_converse.py | 11 ++++++++++ .../pdf_input/test_bedrock_invoke.py | 11 ++++++++++ .../claude_code/pdf_input/test_vertex_ai.py | 11 ++++++++++ .../prompt_caching_1h/test_anthropic.py | 11 ++++++++++ .../prompt_caching_1h/test_azure.py | 11 ++++++++++ .../test_bedrock_converse.py | 11 ++++++++++ .../prompt_caching_1h/test_bedrock_invoke.py | 11 ++++++++++ .../prompt_caching_1h/test_vertex_ai.py | 11 ++++++++++ .../prompt_caching_5m/test_anthropic.py | 11 ++++++++++ .../prompt_caching_5m/test_azure.py | 11 ++++++++++ .../test_bedrock_converse.py | 11 ++++++++++ .../prompt_caching_5m/test_bedrock_invoke.py | 11 ++++++++++ .../prompt_caching_5m/test_vertex_ai.py | 11 ++++++++++ .../structured_outputs/test_anthropic.py | 11 ++++++++++ .../structured_outputs/test_azure.py | 11 ++++++++++ .../test_bedrock_converse.py | 12 ++++++++++ .../structured_outputs/test_bedrock_invoke.py | 12 ++++++++++ .../structured_outputs/test_vertex_ai.py | 12 ++++++++++ .../claude_code/thinking/test_anthropic.py | 12 ++++++++++ tests/e2e/claude_code/thinking/test_azure.py | 12 ++++++++++ .../thinking/test_bedrock_converse.py | 12 ++++++++++ .../thinking/test_bedrock_invoke.py | 12 ++++++++++ .../claude_code/thinking/test_vertex_ai.py | 12 ++++++++++ .../thinking_with_tool_use/test_anthropic.py | 12 ++++++++++ .../thinking_with_tool_use/test_azure.py | 12 ++++++++++ .../test_bedrock_converse.py | 12 ++++++++++ .../test_bedrock_invoke.py | 12 ++++++++++ .../thinking_with_tool_use/test_vertex_ai.py | 12 ++++++++++ .../claude_code/tool_search/test_anthropic.py | 12 ++++++++++ .../e2e/claude_code/tool_search/test_azure.py | 12 ++++++++++ .../tool_search/test_bedrock_converse.py | 12 ++++++++++ .../tool_search/test_bedrock_invoke.py | 22 +++++++++++++++++++ .../claude_code/tool_search/test_vertex_ai.py | 12 ++++++++++ .../claude_code/tool_use/test_anthropic.py | 11 ++++++++++ tests/e2e/claude_code/tool_use/test_azure.py | 11 ++++++++++ .../claude_code/tool_use/test_azure_openai.py | 11 ++++++++++ .../tool_use/test_bedrock_converse.py | 11 ++++++++++ .../tool_use/test_bedrock_invoke.py | 11 ++++++++++ .../tool_use/test_bedrock_mantle.py | 11 ++++++++++ tests/e2e/claude_code/tool_use/test_openai.py | 11 ++++++++++ .../claude_code/tool_use/test_vertex_ai.py | 11 ++++++++++ .../tool_use/test_vertex_ai_gpt.py | 2 ++ .../tool_use_streaming/test_anthropic.py | 11 ++++++++++ .../tool_use_streaming/test_azure.py | 11 ++++++++++ .../tool_use_streaming/test_azure_openai.py | 11 ++++++++++ .../test_bedrock_converse.py | 11 ++++++++++ .../tool_use_streaming/test_bedrock_invoke.py | 11 ++++++++++ .../tool_use_streaming/test_bedrock_mantle.py | 11 ++++++++++ .../tool_use_streaming/test_openai.py | 11 ++++++++++ .../tool_use_streaming/test_vertex_ai.py | 11 ++++++++++ .../tool_use_streaming/test_vertex_ai_gpt.py | 2 ++ .../e2e/claude_code/vision/test_anthropic.py | 11 ++++++++++ tests/e2e/claude_code/vision/test_azure.py | 11 ++++++++++ .../vision/test_bedrock_converse.py | 11 ++++++++++ .../claude_code/vision/test_bedrock_invoke.py | 11 ++++++++++ .../e2e/claude_code/vision/test_vertex_ai.py | 11 ++++++++++ .../claude_code/web_search/test_anthropic.py | 11 ++++++++++ .../e2e/claude_code/web_search/test_azure.py | 11 ++++++++++ .../web_search/test_bedrock_converse.py | 11 ++++++++++ .../web_search/test_bedrock_invoke.py | 11 ++++++++++ .../claude_code/web_search/test_vertex_ai.py | 11 ++++++++++ 100 files changed, 1030 insertions(+) diff --git a/tests/e2e/claude_code/_basic_messaging.py b/tests/e2e/claude_code/_basic_messaging.py index 7c581cc5e38..b207bb6808f 100644 --- a/tests/e2e/claude_code/_basic_messaging.py +++ b/tests/e2e/claude_code/_basic_messaging.py @@ -31,6 +31,8 @@ from typing import Any, Callable, Mapping, Sequence import pytest +from e2e_metadata import step + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -74,6 +76,7 @@ def _count_stream_event_deltas(events: Sequence[Mapping[str, Any]]) -> int: return count +@step("Run Claude Code headless against {models} through the proxy and check every model replies") def run_basic_messaging_cell( *, compat_result, diff --git a/tests/e2e/claude_code/_passthrough.py b/tests/e2e/claude_code/_passthrough.py index be7a475dff7..3693ce25a9c 100644 --- a/tests/e2e/claude_code/_passthrough.py +++ b/tests/e2e/claude_code/_passthrough.py @@ -58,6 +58,8 @@ from typing import Any, Callable, Dict, Mapping, Optional, Sequence import pytest +from e2e_metadata import step + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -118,6 +120,10 @@ def foundry_extra_env(proxy_base_url: str) -> Dict[str, str]: } +@step( + "Run Claude Code headless against {models} through the proxy's native provider passthrough route" + " and check every model replies" +) def run_passthrough_cell( *, compat_result, diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_anthropic.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_anthropic.py index 21383b85da5..1a7e26d37c7 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_anthropic.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_anthropic.py @@ -21,6 +21,7 @@ the matrix builder still sees three rows for this (feature, provider). from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell # Per the PRD: each cell is exercised against three Claude tiers via the @@ -34,6 +35,15 @@ ANTHROPIC_MODELS = [ @pytest.mark.covers("llm.messages.anthropic.basic.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_basic_messaging_non_streaming_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a reply. diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_azure.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_azure.py index 19e88dbe3cb..1fb1cc6fac7 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_azure.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_azure.py @@ -26,6 +26,7 @@ the matrix builder still sees three rows for this (feature, provider). from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell # Per-model aliases registered in the LiteLLM proxy's routing config to @@ -40,6 +41,15 @@ AZURE_MODELS = [ @pytest.mark.covers("llm.messages.azure_foundry.basic.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_basic_messaging_non_streaming_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a reply. diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_azure_openai.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_azure_openai.py index 77876c8f7ee..2b90b4e9f02 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_azure_openai.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_azure_openai.py @@ -25,6 +25,7 @@ green if all three pass. from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell AZURE_OPENAI_MODELS = [ @@ -34,6 +35,15 @@ AZURE_OPENAI_MODELS = [ ] +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE,), + models=tuple(AZURE_OPENAI_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_basic_messaging_non_streaming_azure_openai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty reply from each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py index 2b0f49bc205..4b02f10e6a9 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py @@ -21,6 +21,7 @@ the matrix builder still sees three rows for this (feature, provider). from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell # Per-model aliases registered in the LiteLLM proxy's routing config to @@ -35,6 +36,15 @@ BEDROCK_CONVERSE_MODELS = [ @pytest.mark.covers("llm.messages.bedrock_converse.basic.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_basic_messaging_non_streaming_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a reply.""" run_basic_messaging_cell( diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py index 937ea5ee27e..874dcebdc35 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py @@ -21,6 +21,7 @@ the matrix builder still sees three rows for this (feature, provider). from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell # Per-model aliases registered in the LiteLLM proxy's routing config to @@ -35,6 +36,15 @@ BEDROCK_INVOKE_MODELS = [ @pytest.mark.covers("llm.messages.bedrock_invoke.basic.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_basic_messaging_non_streaming_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a reply.""" run_basic_messaging_cell( diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_mantle.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_mantle.py index 8a64547a732..28e723bb450 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_mantle.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_mantle.py @@ -25,6 +25,7 @@ COMPAT_MANTLE_CELLS=1 (see `claude_code._gpt_cells`). from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell from claude_code._gpt_cells import skip_unless_mantle_cells_enabled @@ -35,6 +36,15 @@ BEDROCK_MANTLE_MODELS = [ ] +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK_MANTLE,), + models=tuple(BEDROCK_MANTLE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_basic_messaging_non_streaming_bedrock_mantle(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty reply from each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_openai.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_openai.py index 323c2f11173..27ab244c52a 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_openai.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_openai.py @@ -22,6 +22,7 @@ green if all three pass. from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell from claude_code._gpt_cells import skip_unless_openai_gpt_cells_enabled @@ -32,6 +33,15 @@ OPENAI_MODELS = [ ] +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.OPENAI,), + models=tuple(OPENAI_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_basic_messaging_non_streaming_openai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty reply from each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai.py index c46e5a8f762..b0b41dc3f87 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai.py @@ -21,6 +21,7 @@ the matrix builder still sees three rows for this (feature, provider). from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell # Per-model aliases registered in the LiteLLM proxy's routing config to @@ -35,6 +36,15 @@ VERTEX_AI_MODELS = [ @pytest.mark.covers("llm.messages.vertex.basic.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_basic_messaging_non_streaming_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a reply.""" run_basic_messaging_cell( diff --git a/tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai_gpt.py b/tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai_gpt.py index 3b155b6ac9d..7e8358a7975 100644 --- a/tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai_gpt.py +++ b/tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai_gpt.py @@ -16,9 +16,11 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations +from e2e_metadata import Domain, Subject, meta from claude_code._gpt_cells import VERTEX_AI_GPT_NOT_APPLICABLE_REASON +@meta(Subject(domain=Domain.LLM_TRANSLATION)) def test_basic_messaging_non_streaming_vertex_ai_gpt(compat_result): """Record the static not_applicable outcome for this cell.""" compat_result.set( diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_anthropic.py b/tests/e2e/claude_code/basic_messaging_streaming/test_anthropic.py index ce453f3e523..69ca0ee475a 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_anthropic.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_anthropic.py @@ -26,6 +26,7 @@ sees three rows for this (feature, provider). from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell ANTHROPIC_MODELS = [ @@ -36,6 +37,15 @@ ANTHROPIC_MODELS = [ @pytest.mark.covers("llm.messages.anthropic.basic.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + mode=Mode.STREAM, + ) +) def test_basic_messaging_streaming_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty streamed reply (one row per Claude tier). diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_azure.py b/tests/e2e/claude_code/basic_messaging_streaming/test_azure.py index 3307194e862..f9fc39c8f65 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_azure.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_azure.py @@ -20,6 +20,7 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell AZURE_MODELS = [ @@ -30,6 +31,15 @@ AZURE_MODELS = [ @pytest.mark.covers("llm.messages.azure_foundry.basic.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + mode=Mode.STREAM, + ) +) def test_basic_messaging_streaming_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty streamed reply (one row per Claude tier). diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_azure_openai.py b/tests/e2e/claude_code/basic_messaging_streaming/test_azure_openai.py index 357596590c7..d1ff8578f09 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_azure_openai.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_azure_openai.py @@ -24,6 +24,7 @@ green if all three pass. from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell AZURE_OPENAI_MODELS = [ @@ -33,6 +34,15 @@ AZURE_OPENAI_MODELS = [ ] +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE,), + models=tuple(AZURE_OPENAI_MODELS), + mode=Mode.STREAM, + ) +) def test_basic_messaging_streaming_azure_openai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty streamed reply from each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_converse.py b/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_converse.py index a8bc0b77a5d..3ad0b930df4 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_converse.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_converse.py @@ -16,6 +16,7 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell BEDROCK_CONVERSE_MODELS = [ @@ -26,6 +27,15 @@ BEDROCK_CONVERSE_MODELS = [ @pytest.mark.covers("llm.messages.bedrock_converse.basic.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + mode=Mode.STREAM, + ) +) def test_basic_messaging_streaming_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty streamed reply (one row per Claude tier). diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_invoke.py b/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_invoke.py index c0ece0e0721..1de3236d2aa 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_invoke.py @@ -16,6 +16,7 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell BEDROCK_INVOKE_MODELS = [ @@ -26,6 +27,15 @@ BEDROCK_INVOKE_MODELS = [ @pytest.mark.covers("llm.messages.bedrock_invoke.basic.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + mode=Mode.STREAM, + ) +) def test_basic_messaging_streaming_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty streamed reply (one row per Claude tier). diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_mantle.py b/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_mantle.py index 38297e6a3e5..679731003bb 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_mantle.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_mantle.py @@ -25,6 +25,7 @@ COMPAT_MANTLE_CELLS=1 (see `claude_code._gpt_cells`). from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell from claude_code._gpt_cells import skip_unless_mantle_cells_enabled @@ -35,6 +36,15 @@ BEDROCK_MANTLE_MODELS = [ ] +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK_MANTLE,), + models=tuple(BEDROCK_MANTLE_MODELS), + mode=Mode.STREAM, + ) +) def test_basic_messaging_streaming_bedrock_mantle(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty streamed reply from each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_openai.py b/tests/e2e/claude_code/basic_messaging_streaming/test_openai.py index a7945fb92c0..c1617253f39 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_openai.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_openai.py @@ -24,6 +24,7 @@ green if all three pass. from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell from claude_code._gpt_cells import skip_unless_openai_gpt_cells_enabled @@ -34,6 +35,15 @@ OPENAI_MODELS = [ ] +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.OPENAI,), + models=tuple(OPENAI_MODELS), + mode=Mode.STREAM, + ) +) def test_basic_messaging_streaming_openai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty streamed reply from each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai.py b/tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai.py index 13f1a0abf40..1201c71dce6 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai.py @@ -16,6 +16,7 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._basic_messaging import run_basic_messaging_cell VERTEX_AI_MODELS = [ @@ -26,6 +27,15 @@ VERTEX_AI_MODELS = [ @pytest.mark.covers("llm.messages.vertex.basic.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + mode=Mode.STREAM, + ) +) def test_basic_messaging_streaming_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a non-empty streamed reply (one row per Claude tier). diff --git a/tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai_gpt.py b/tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai_gpt.py index f6aa01de521..6a6cec8411a 100644 --- a/tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai_gpt.py +++ b/tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai_gpt.py @@ -16,9 +16,11 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations +from e2e_metadata import Domain, Subject, meta from claude_code._gpt_cells import VERTEX_AI_GPT_NOT_APPLICABLE_REASON +@meta(Subject(domain=Domain.LLM_TRANSLATION)) def test_basic_messaging_streaming_vertex_ai_gpt(compat_result): """Record the static not_applicable outcome for this cell.""" compat_result.set( diff --git a/tests/e2e/claude_code/cli_driver.py b/tests/e2e/claude_code/cli_driver.py index 6996849de80..6f7a7b39a39 100644 --- a/tests/e2e/claude_code/cli_driver.py +++ b/tests/e2e/claude_code/cli_driver.py @@ -25,6 +25,8 @@ from concurrent.futures import ThreadPoolExecutor, as_completed from dataclasses import dataclass, field from typing import Any, Callable, Dict, List, Mapping, Optional, Sequence, Tuple, Union +from e2e_metadata import step + from claude_code.rate_limiter import ( RateLimiter, get_default_limiter, @@ -211,6 +213,7 @@ class DriverResult: duration_ms: Optional[int] = None +@step("Run Claude Code headless against {model} through the proxy") def run_claude( *, prompt: Optional[str], @@ -395,6 +398,7 @@ def _matches_failure_shape(outcome: ModelResult, pattern: "re.Pattern[str]") -> return bool(pattern.search(failure_diagnostic(outcome))) +@step("Run Claude Code headless against {models} in parallel through the proxy") def run_claude_models_parallel( *, models: Sequence[str], diff --git a/tests/e2e/claude_code/count_tokens/test_anthropic.py b/tests/e2e/claude_code/count_tokens/test_anthropic.py index 05110d24e86..19e7bcfbc81 100644 --- a/tests/e2e/claude_code/count_tokens/test_anthropic.py +++ b/tests/e2e/claude_code/count_tokens/test_anthropic.py @@ -39,6 +39,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_count_tokens_shape, @@ -54,6 +55,15 @@ ANTHROPIC_MODELS = [ @pytest.mark.covers("llm.messages.anthropic.count_tokens.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.COUNT_TOKENS, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_count_tokens_anthropic(compat_result): """Probe `/v1/messages/count_tokens` for each Anthropic tier and assert the response shape.""" diff --git a/tests/e2e/claude_code/count_tokens/test_azure.py b/tests/e2e/claude_code/count_tokens/test_azure.py index c60c623ae89..256babf3cd7 100644 --- a/tests/e2e/claude_code/count_tokens/test_azure.py +++ b/tests/e2e/claude_code/count_tokens/test_azure.py @@ -39,6 +39,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_count_tokens_shape, @@ -54,6 +55,15 @@ AZURE_MODELS = [ @pytest.mark.covers("llm.messages.azure_foundry.count_tokens.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.COUNT_TOKENS, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_count_tokens_azure(compat_result): """Probe `/v1/messages/count_tokens` for each Azure (Microsoft Foundry) tier and assert the response shape.""" diff --git a/tests/e2e/claude_code/count_tokens/test_bedrock_converse.py b/tests/e2e/claude_code/count_tokens/test_bedrock_converse.py index 0cb4766bb31..de5f4eae3ee 100644 --- a/tests/e2e/claude_code/count_tokens/test_bedrock_converse.py +++ b/tests/e2e/claude_code/count_tokens/test_bedrock_converse.py @@ -39,6 +39,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_count_tokens_shape, @@ -54,6 +55,15 @@ BEDROCK_CONVERSE_MODELS = [ @pytest.mark.covers("llm.messages.bedrock_converse.count_tokens.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.COUNT_TOKENS, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_count_tokens_bedrock_converse(compat_result): """Probe `/v1/messages/count_tokens` for each Bedrock (Converse) tier and assert the response shape.""" diff --git a/tests/e2e/claude_code/count_tokens/test_bedrock_invoke.py b/tests/e2e/claude_code/count_tokens/test_bedrock_invoke.py index f1389574527..56dea8a5ad6 100644 --- a/tests/e2e/claude_code/count_tokens/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/count_tokens/test_bedrock_invoke.py @@ -39,6 +39,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_count_tokens_shape, @@ -54,6 +55,15 @@ BEDROCK_INVOKE_MODELS = [ @pytest.mark.covers("llm.messages.bedrock_invoke.count_tokens.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.COUNT_TOKENS, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_count_tokens_bedrock_invoke(compat_result): """Probe `/v1/messages/count_tokens` for each Bedrock (Invoke) tier and assert the response shape.""" diff --git a/tests/e2e/claude_code/count_tokens/test_vertex_ai.py b/tests/e2e/claude_code/count_tokens/test_vertex_ai.py index 0894214d4f0..63d12fc5dae 100644 --- a/tests/e2e/claude_code/count_tokens/test_vertex_ai.py +++ b/tests/e2e/claude_code/count_tokens/test_vertex_ai.py @@ -39,6 +39,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_count_tokens_shape, @@ -55,6 +56,15 @@ VERTEX_AI_MODELS = [ @pytest.mark.skip(reason="stage red: Vertex returns not supported for token counting for Claude aliases") @pytest.mark.covers("llm.messages.vertex.count_tokens.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.COUNT_TOKENS, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_count_tokens_vertex_ai(compat_result): """Probe `/v1/messages/count_tokens` for each Vertex AI tier and assert the response shape.""" diff --git a/tests/e2e/claude_code/http_probe.py b/tests/e2e/claude_code/http_probe.py index 8aba54576c4..a6c587f98cf 100644 --- a/tests/e2e/claude_code/http_probe.py +++ b/tests/e2e/claude_code/http_probe.py @@ -42,6 +42,7 @@ from e2e_http import ( UnknownApiError, ValidationError, ) +from e2e_metadata import step from models import ( AnthropicAssistantTurn, AnthropicCustomTool, @@ -109,6 +110,7 @@ def _acquire(model: str, rate_limiter: RateLimiter | None) -> None: limiter.acquire(infer_provider(model)) +@step('Count tokens with /v1/messages/count_tokens for {model} on the message "{message}"') def probe_count_tokens( *, client: ProxyClient, @@ -132,6 +134,7 @@ def probe_count_tokens( ) +@step("Send a /v1/messages request to {model} with the tool_search tool declared") def probe_tool_search( *, client: ProxyClient, @@ -228,6 +231,10 @@ def _replay_history(answer: AnthropicMessagesResponse) -> tuple[AnthropicMessage ) +@step( + "Send a /v1/messages request to {model} with the tool_search tool declared," + " then send its answer back as history in a second request" +) def probe_tool_search_multiturn( *, client: ProxyClient, diff --git a/tests/e2e/claude_code/long_context_1m/test_anthropic.py b/tests/e2e/claude_code/long_context_1m/test_anthropic.py index 0f53e512ace..1de515c11f0 100644 --- a/tests/e2e/claude_code/long_context_1m/test_anthropic.py +++ b/tests/e2e/claude_code/long_context_1m/test_anthropic.py @@ -58,6 +58,7 @@ from typing import Sequence import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -155,6 +156,15 @@ def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: @pytest.mark.skip(reason="stage red: 1M long_context not green on stage Anthropic path yet (200k sonnet / model alias)") @pytest.mark.covers("llm.messages.anthropic.long_context_1m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_long_context_1m_anthropic(compat_result): """Drive the `claude` CLI with a ~210k-token prompt and the `context-1m-2025-08-07` beta header; assert no 400 / 413 and a diff --git a/tests/e2e/claude_code/long_context_1m/test_azure.py b/tests/e2e/claude_code/long_context_1m/test_azure.py index cdaa7f08178..eaad5e3de9f 100644 --- a/tests/e2e/claude_code/long_context_1m/test_azure.py +++ b/tests/e2e/claude_code/long_context_1m/test_azure.py @@ -58,6 +58,7 @@ from typing import Sequence import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -155,6 +156,15 @@ def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: @pytest.mark.skip(reason="stage red: 1M long_context not green on stage Azure Foundry deployments yet") @pytest.mark.covers("llm.messages.azure_foundry.long_context_1m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_long_context_1m_azure(compat_result): """Drive the `claude` CLI (Azure (Microsoft Foundry)) with a ~210k-token prompt and the `context-1m-2025-08-07` beta header; assert no 400 / 413 and a diff --git a/tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py b/tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py index 38aeef2ae63..eb72a36bf14 100644 --- a/tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py +++ b/tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py @@ -58,6 +58,7 @@ from typing import Sequence import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -155,6 +156,15 @@ def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: @pytest.mark.skip(reason="stage red: 1M long_context not green on stage Bedrock Converse deployments yet") @pytest.mark.covers("llm.messages.bedrock_converse.long_context_1m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_long_context_1m_bedrock_converse(compat_result): """Drive the `claude` CLI (Bedrock (Converse)) with a ~210k-token prompt and the `context-1m-2025-08-07` beta header; assert no 400 / 413 and a diff --git a/tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py b/tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py index f652af4aa22..dd960c15428 100644 --- a/tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py @@ -58,6 +58,7 @@ from typing import Sequence import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -155,6 +156,15 @@ def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: @pytest.mark.skip(reason="stage red: 1M long_context not green on stage Bedrock Invoke deployments yet") @pytest.mark.covers("llm.messages.bedrock_invoke.long_context_1m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_long_context_1m_bedrock_invoke(compat_result): """Drive the `claude` CLI (Bedrock (Invoke)) with a ~210k-token prompt and the `context-1m-2025-08-07` beta header; assert no 400 / 413 and a diff --git a/tests/e2e/claude_code/long_context_1m/test_vertex_ai.py b/tests/e2e/claude_code/long_context_1m/test_vertex_ai.py index 0ad68aac138..1f344d80ba8 100644 --- a/tests/e2e/claude_code/long_context_1m/test_vertex_ai.py +++ b/tests/e2e/claude_code/long_context_1m/test_vertex_ai.py @@ -58,6 +58,7 @@ from typing import Sequence import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -155,6 +156,15 @@ def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: @pytest.mark.skip(reason="stage red: 1M long_context not green on stage Vertex deployments yet") @pytest.mark.covers("llm.messages.vertex.long_context_1m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + mode=Mode.NONSTREAM, + ) +) def test_long_context_1m_vertex_ai(compat_result): """Drive the `claude` CLI (Vertex AI) with a ~210k-token prompt and the `context-1m-2025-08-07` beta header; assert no 400 / 413 and a diff --git a/tests/e2e/claude_code/passthrough/test_anthropic.py b/tests/e2e/claude_code/passthrough/test_anthropic.py index 8382342ae12..bf6ab0524d0 100644 --- a/tests/e2e/claude_code/passthrough/test_anthropic.py +++ b/tests/e2e/claude_code/passthrough/test_anthropic.py @@ -22,6 +22,7 @@ no per-provider transformation is involved. from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._passthrough import ( ANTHROPIC_PASSTHROUGH_BASE_PATH, run_passthrough_cell, @@ -34,6 +35,15 @@ ANTHROPIC_MODELS = [ ] +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + mode=Mode.STREAM, + ) +) def test_passthrough_anthropic(compat_result): """Drive the `claude` CLI through `{proxy}/anthropic` and assert a reply.""" run_passthrough_cell( diff --git a/tests/e2e/claude_code/passthrough/test_azure.py b/tests/e2e/claude_code/passthrough/test_azure.py index 7365b4f50da..d2690025f08 100644 --- a/tests/e2e/claude_code/passthrough/test_azure.py +++ b/tests/e2e/claude_code/passthrough/test_azure.py @@ -42,6 +42,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._passthrough import foundry_extra_env, run_passthrough_cell AZURE_MODELS = [ @@ -52,6 +53,15 @@ AZURE_MODELS = [ @pytest.mark.skip(reason="stage red: /azure passthrough drops client headers (e.g. anthropic-version); product gap") +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.AZURE,), + models=tuple(AZURE_MODELS), + mode=Mode.STREAM, + ) +) def test_passthrough_azure(compat_result): """Drive the `claude` CLI through `{proxy}/azure` and assert a reply.""" run_passthrough_cell( diff --git a/tests/e2e/claude_code/passthrough/test_bedrock_converse.py b/tests/e2e/claude_code/passthrough/test_bedrock_converse.py index d1093a7a958..e1605be9e49 100644 --- a/tests/e2e/claude_code/passthrough/test_bedrock_converse.py +++ b/tests/e2e/claude_code/passthrough/test_bedrock_converse.py @@ -18,7 +18,10 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations +from e2e_metadata import Domain, Subject, meta + +@meta(Subject(domain=Domain.PASSTHROUGH)) def test_passthrough_bedrock_converse(compat_result): """Report not_applicable: Claude Code has no Converse-wire mode.""" compat_result.set( diff --git a/tests/e2e/claude_code/passthrough/test_bedrock_invoke.py b/tests/e2e/claude_code/passthrough/test_bedrock_invoke.py index f1f28ab5b4c..d95118991fa 100644 --- a/tests/e2e/claude_code/passthrough/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/passthrough/test_bedrock_invoke.py @@ -23,6 +23,7 @@ cell. from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._passthrough import bedrock_extra_env, run_passthrough_cell BEDROCK_INVOKE_MODELS = [ @@ -32,6 +33,15 @@ BEDROCK_INVOKE_MODELS = [ ] +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + mode=Mode.STREAM, + ) +) def test_passthrough_bedrock_invoke(compat_result): """Drive the `claude` CLI through `{proxy}/bedrock` and assert a reply.""" run_passthrough_cell( diff --git a/tests/e2e/claude_code/passthrough/test_vertex_ai.py b/tests/e2e/claude_code/passthrough/test_vertex_ai.py index 790f8b60c8f..3f84cdf02fa 100644 --- a/tests/e2e/claude_code/passthrough/test_vertex_ai.py +++ b/tests/e2e/claude_code/passthrough/test_vertex_ai.py @@ -26,6 +26,7 @@ Google and every tier fails with a 401. from __future__ import annotations +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from claude_code._passthrough import run_passthrough_cell, vertex_extra_env VERTEX_MODELS = [ @@ -35,6 +36,15 @@ VERTEX_MODELS = [ ] +@meta( + Subject( + domain=Domain.PASSTHROUGH, + route=Route.PASSTHROUGH, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_MODELS), + mode=Mode.STREAM, + ) +) def test_passthrough_vertex_ai(compat_result): """Drive the `claude` CLI through `{proxy}/vertex_ai` and assert a reply.""" run_passthrough_cell( diff --git a/tests/e2e/claude_code/pdf_input/test_anthropic.py b/tests/e2e/claude_code/pdf_input/test_anthropic.py index 21c8028ef1c..59a91036f26 100644 --- a/tests/e2e/claude_code/pdf_input/test_anthropic.py +++ b/tests/e2e/claude_code/pdf_input/test_anthropic.py @@ -24,6 +24,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -106,6 +107,16 @@ def _build_minimal_pdf(marker: str) -> bytes: @pytest.mark.covers("llm.messages.anthropic.pdf_input.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.PDF_INPUT,), + mode=Mode.NONSTREAM, + ) +) def test_pdf_input_anthropic(compat_result, tmp_path): """Drive the `claude` CLI against the LiteLLM proxy with a PDF attached via the Read tool and assert the reply references it.""" diff --git a/tests/e2e/claude_code/pdf_input/test_azure.py b/tests/e2e/claude_code/pdf_input/test_azure.py index 34ae3732b99..3f7642321cf 100644 --- a/tests/e2e/claude_code/pdf_input/test_azure.py +++ b/tests/e2e/claude_code/pdf_input/test_azure.py @@ -17,6 +17,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -83,6 +84,16 @@ def _build_minimal_pdf(marker: str) -> bytes: @pytest.mark.covers("llm.messages.azure_foundry.pdf_input.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.PDF_INPUT,), + mode=Mode.NONSTREAM, + ) +) def test_pdf_input_azure(compat_result, tmp_path): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/pdf_input/test_bedrock_converse.py b/tests/e2e/claude_code/pdf_input/test_bedrock_converse.py index 76aa84f0f47..14020799c21 100644 --- a/tests/e2e/claude_code/pdf_input/test_bedrock_converse.py +++ b/tests/e2e/claude_code/pdf_input/test_bedrock_converse.py @@ -23,6 +23,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -89,6 +90,16 @@ def _build_minimal_pdf(marker: str) -> bytes: @pytest.mark.covers("llm.messages.bedrock_converse.pdf_input.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.PDF_INPUT,), + mode=Mode.NONSTREAM, + ) +) def test_pdf_input_bedrock_converse(compat_result, tmp_path): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/pdf_input/test_bedrock_invoke.py b/tests/e2e/claude_code/pdf_input/test_bedrock_invoke.py index 4450266bb6b..5b3ba208421 100644 --- a/tests/e2e/claude_code/pdf_input/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/pdf_input/test_bedrock_invoke.py @@ -22,6 +22,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -88,6 +89,16 @@ def _build_minimal_pdf(marker: str) -> bytes: @pytest.mark.covers("llm.messages.bedrock_invoke.pdf_input.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.PDF_INPUT,), + mode=Mode.NONSTREAM, + ) +) def test_pdf_input_bedrock_invoke(compat_result, tmp_path): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/pdf_input/test_vertex_ai.py b/tests/e2e/claude_code/pdf_input/test_vertex_ai.py index b78f58cfda1..5f42c00efcb 100644 --- a/tests/e2e/claude_code/pdf_input/test_vertex_ai.py +++ b/tests/e2e/claude_code/pdf_input/test_vertex_ai.py @@ -17,6 +17,7 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -83,6 +84,16 @@ def _build_minimal_pdf(marker: str) -> bytes: @pytest.mark.covers("llm.messages.vertex.pdf_input.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.PDF_INPUT,), + mode=Mode.NONSTREAM, + ) +) def test_pdf_input_vertex_ai(compat_result, tmp_path): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/prompt_caching_1h/test_anthropic.py b/tests/e2e/claude_code/prompt_caching_1h/test_anthropic.py index 637be1c551d..60045eff8b9 100644 --- a/tests/e2e/claude_code/prompt_caching_1h/test_anthropic.py +++ b/tests/e2e/claude_code/prompt_caching_1h/test_anthropic.py @@ -28,6 +28,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -64,6 +65,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.anthropic.prompt_cache_1h.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_1h_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with the 1h TTL opt-in env var set, and assert the upstream usage block diff --git a/tests/e2e/claude_code/prompt_caching_1h/test_azure.py b/tests/e2e/claude_code/prompt_caching_1h/test_azure.py index f34557b3c5f..86dc4c8e50a 100644 --- a/tests/e2e/claude_code/prompt_caching_1h/test_azure.py +++ b/tests/e2e/claude_code/prompt_caching_1h/test_azure.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -48,6 +49,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.azure_foundry.prompt_cache_1h.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_1h_azure(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/prompt_caching_1h/test_bedrock_converse.py b/tests/e2e/claude_code/prompt_caching_1h/test_bedrock_converse.py index bf62a49444c..d1106ca4350 100644 --- a/tests/e2e/claude_code/prompt_caching_1h/test_bedrock_converse.py +++ b/tests/e2e/claude_code/prompt_caching_1h/test_bedrock_converse.py @@ -24,6 +24,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -56,6 +57,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.bedrock_converse.prompt_cache_1h.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_1h_bedrock_converse(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/prompt_caching_1h/test_bedrock_invoke.py b/tests/e2e/claude_code/prompt_caching_1h/test_bedrock_invoke.py index dc3468702d4..9c1166eb3b6 100644 --- a/tests/e2e/claude_code/prompt_caching_1h/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/prompt_caching_1h/test_bedrock_invoke.py @@ -26,6 +26,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -60,6 +61,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.bedrock_invoke.prompt_cache_1h.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_1h_bedrock_invoke(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/prompt_caching_1h/test_vertex_ai.py b/tests/e2e/claude_code/prompt_caching_1h/test_vertex_ai.py index 66cf961fcfc..89b6d7e1d4e 100644 --- a/tests/e2e/claude_code/prompt_caching_1h/test_vertex_ai.py +++ b/tests/e2e/claude_code/prompt_caching_1h/test_vertex_ai.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -48,6 +49,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.vertex.prompt_cache_1h.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_1h_vertex_ai(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/prompt_caching_5m/test_anthropic.py b/tests/e2e/claude_code/prompt_caching_5m/test_anthropic.py index ef551beb45c..c97df9afe2c 100644 --- a/tests/e2e/claude_code/prompt_caching_5m/test_anthropic.py +++ b/tests/e2e/claude_code/prompt_caching_5m/test_anthropic.py @@ -26,6 +26,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -55,6 +56,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.anthropic.prompt_cache_5m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_5m_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream usage block surfaces a non-zero cache token count.""" diff --git a/tests/e2e/claude_code/prompt_caching_5m/test_azure.py b/tests/e2e/claude_code/prompt_caching_5m/test_azure.py index 9d4137e0726..7705af62ab4 100644 --- a/tests/e2e/claude_code/prompt_caching_5m/test_azure.py +++ b/tests/e2e/claude_code/prompt_caching_5m/test_azure.py @@ -26,6 +26,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -53,6 +54,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.azure_foundry.prompt_cache_5m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_5m_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream usage block surfaces a non-zero cache token count.""" diff --git a/tests/e2e/claude_code/prompt_caching_5m/test_bedrock_converse.py b/tests/e2e/claude_code/prompt_caching_5m/test_bedrock_converse.py index c9b34c010b0..4a5a7dfd660 100644 --- a/tests/e2e/claude_code/prompt_caching_5m/test_bedrock_converse.py +++ b/tests/e2e/claude_code/prompt_caching_5m/test_bedrock_converse.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -46,6 +47,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.bedrock_converse.prompt_cache_5m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_5m_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream usage block surfaces a non-zero cache token count.""" diff --git a/tests/e2e/claude_code/prompt_caching_5m/test_bedrock_invoke.py b/tests/e2e/claude_code/prompt_caching_5m/test_bedrock_invoke.py index b95c509ba3c..bbe483242bd 100644 --- a/tests/e2e/claude_code/prompt_caching_5m/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/prompt_caching_5m/test_bedrock_invoke.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -46,6 +47,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.bedrock_invoke.prompt_cache_5m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_5m_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream usage block surfaces a non-zero cache token count.""" diff --git a/tests/e2e/claude_code/prompt_caching_5m/test_vertex_ai.py b/tests/e2e/claude_code/prompt_caching_5m/test_vertex_ai.py index f79377b7372..3f325efcc93 100644 --- a/tests/e2e/claude_code/prompt_caching_5m/test_vertex_ai.py +++ b/tests/e2e/claude_code/prompt_caching_5m/test_vertex_ai.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Optional import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -46,6 +47,16 @@ def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.vertex.prompt_cache_5m.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) +) def test_prompt_caching_5m_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream usage block surfaces a non-zero cache token count.""" diff --git a/tests/e2e/claude_code/structured_outputs/test_anthropic.py b/tests/e2e/claude_code/structured_outputs/test_anthropic.py index 3dc4c7ab8f2..c8cb7e7386c 100644 --- a/tests/e2e/claude_code/structured_outputs/test_anthropic.py +++ b/tests/e2e/claude_code/structured_outputs/test_anthropic.py @@ -53,6 +53,7 @@ from typing import Any, Mapping, Optional, Sequence, Tuple import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -151,6 +152,16 @@ def _validate_against_schema( @pytest.mark.covers("llm.messages.anthropic.structured_output.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) +) def test_structured_outputs_anthropic(compat_result): """Drive `claude --json-schema ...` against the LiteLLM proxy and assert the trailing `result` event contains a schema-conforming diff --git a/tests/e2e/claude_code/structured_outputs/test_azure.py b/tests/e2e/claude_code/structured_outputs/test_azure.py index 7a776ed55ad..0708920f5e7 100644 --- a/tests/e2e/claude_code/structured_outputs/test_azure.py +++ b/tests/e2e/claude_code/structured_outputs/test_azure.py @@ -53,6 +53,7 @@ from typing import Any, Mapping, Optional, Sequence, Tuple import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -151,6 +152,16 @@ def _validate_against_schema( @pytest.mark.covers("llm.messages.azure_foundry.structured_output.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) +) def test_structured_outputs_azure(compat_result): """Drive `claude --json-schema ...` against the LiteLLM proxy and assert the trailing `result` event contains a schema-conforming diff --git a/tests/e2e/claude_code/structured_outputs/test_bedrock_converse.py b/tests/e2e/claude_code/structured_outputs/test_bedrock_converse.py index 345d7c327cf..7228c194f85 100644 --- a/tests/e2e/claude_code/structured_outputs/test_bedrock_converse.py +++ b/tests/e2e/claude_code/structured_outputs/test_bedrock_converse.py @@ -53,6 +53,8 @@ from typing import Any, Mapping, Optional, Sequence, Tuple import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -151,6 +153,16 @@ def _validate_against_schema( @pytest.mark.covers("llm.messages.bedrock_converse.structured_output.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) +) def test_structured_outputs_bedrock_converse(compat_result): """Drive `claude --json-schema ...` against the LiteLLM proxy and assert the trailing `result` event contains a schema-conforming diff --git a/tests/e2e/claude_code/structured_outputs/test_bedrock_invoke.py b/tests/e2e/claude_code/structured_outputs/test_bedrock_invoke.py index 0cf48c72d4f..c415de9e396 100644 --- a/tests/e2e/claude_code/structured_outputs/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/structured_outputs/test_bedrock_invoke.py @@ -53,6 +53,8 @@ from typing import Any, Mapping, Optional, Sequence, Tuple import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -151,6 +153,16 @@ def _validate_against_schema( @pytest.mark.covers("llm.messages.bedrock_invoke.structured_output.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) +) def test_structured_outputs_bedrock_invoke(compat_result): """Drive `claude --json-schema ...` against the LiteLLM proxy and assert the trailing `result` event contains a schema-conforming diff --git a/tests/e2e/claude_code/structured_outputs/test_vertex_ai.py b/tests/e2e/claude_code/structured_outputs/test_vertex_ai.py index 24f5a0c35d4..a4cf669c19f 100644 --- a/tests/e2e/claude_code/structured_outputs/test_vertex_ai.py +++ b/tests/e2e/claude_code/structured_outputs/test_vertex_ai.py @@ -53,6 +53,8 @@ from typing import Any, Mapping, Optional, Sequence, Tuple import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -151,6 +153,16 @@ def _validate_against_schema( @pytest.mark.covers("llm.messages.vertex.structured_output.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.RESPONSE_SCHEMA,), + mode=Mode.NONSTREAM, + ) +) def test_structured_outputs_vertex_ai(compat_result): """Drive `claude --json-schema ...` against the LiteLLM proxy and assert the trailing `result` event contains a schema-conforming diff --git a/tests/e2e/claude_code/thinking/test_anthropic.py b/tests/e2e/claude_code/thinking/test_anthropic.py index ebb2445fb6d..2c3a8f0a8a9 100644 --- a/tests/e2e/claude_code/thinking/test_anthropic.py +++ b/tests/e2e/claude_code/thinking/test_anthropic.py @@ -24,6 +24,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -75,6 +77,16 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.anthropic.thinking.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" diff --git a/tests/e2e/claude_code/thinking/test_azure.py b/tests/e2e/claude_code/thinking/test_azure.py index ffd5ca92df0..0828dd033f4 100644 --- a/tests/e2e/claude_code/thinking/test_azure.py +++ b/tests/e2e/claude_code/thinking/test_azure.py @@ -27,6 +27,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -63,6 +65,16 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.azure_foundry.thinking.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" diff --git a/tests/e2e/claude_code/thinking/test_bedrock_converse.py b/tests/e2e/claude_code/thinking/test_bedrock_converse.py index 0b409f18ea7..fd075298234 100644 --- a/tests/e2e/claude_code/thinking/test_bedrock_converse.py +++ b/tests/e2e/claude_code/thinking/test_bedrock_converse.py @@ -19,6 +19,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -55,6 +57,16 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.bedrock_converse.thinking.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" diff --git a/tests/e2e/claude_code/thinking/test_bedrock_invoke.py b/tests/e2e/claude_code/thinking/test_bedrock_invoke.py index a2c97eae321..c115ba9e408 100644 --- a/tests/e2e/claude_code/thinking/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/thinking/test_bedrock_invoke.py @@ -19,6 +19,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -55,6 +57,16 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.bedrock_invoke.thinking.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" diff --git a/tests/e2e/claude_code/thinking/test_vertex_ai.py b/tests/e2e/claude_code/thinking/test_vertex_ai.py index f1a1c5b6cee..b947f97e7ac 100644 --- a/tests/e2e/claude_code/thinking/test_vertex_ai.py +++ b/tests/e2e/claude_code/thinking/test_vertex_ai.py @@ -19,6 +19,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -55,6 +57,16 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.vertex.thinking.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.REASONING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" diff --git a/tests/e2e/claude_code/thinking_with_tool_use/test_anthropic.py b/tests/e2e/claude_code/thinking_with_tool_use/test_anthropic.py index 7e39ea26d42..8ddbdba04f9 100644 --- a/tests/e2e/claude_code/thinking_with_tool_use/test_anthropic.py +++ b/tests/e2e/claude_code/thinking_with_tool_use/test_anthropic.py @@ -28,6 +28,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -89,6 +91,16 @@ def _has_block_type( @pytest.mark.covers("llm.messages.anthropic.thinking_with_tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.REASONING, Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_with_tool_use_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and tool use, and assert both `thinking` and `tool_use` diff --git a/tests/e2e/claude_code/thinking_with_tool_use/test_azure.py b/tests/e2e/claude_code/thinking_with_tool_use/test_azure.py index 0371a10f8a6..6dbd6b27fde 100644 --- a/tests/e2e/claude_code/thinking_with_tool_use/test_azure.py +++ b/tests/e2e/claude_code/thinking_with_tool_use/test_azure.py @@ -22,6 +22,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -70,6 +72,16 @@ def _has_block_type( @pytest.mark.covers("llm.messages.azure_foundry.thinking_with_tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.REASONING, Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_with_tool_use_azure(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_converse.py b/tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_converse.py index 026d2a3707f..b3869f157b5 100644 --- a/tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_converse.py +++ b/tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_converse.py @@ -27,6 +27,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -75,6 +77,16 @@ def _has_block_type( @pytest.mark.covers("llm.messages.bedrock_converse.thinking_with_tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.REASONING, Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_with_tool_use_bedrock_converse(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_invoke.py b/tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_invoke.py index 1dd4cf0a73c..384ce0bb3c6 100644 --- a/tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_invoke.py @@ -29,6 +29,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -77,6 +79,16 @@ def _has_block_type( @pytest.mark.covers("llm.messages.bedrock_invoke.thinking_with_tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.REASONING, Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_with_tool_use_bedrock_invoke(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/thinking_with_tool_use/test_vertex_ai.py b/tests/e2e/claude_code/thinking_with_tool_use/test_vertex_ai.py index b25228edb55..c3a3dd58533 100644 --- a/tests/e2e/claude_code/thinking_with_tool_use/test_vertex_ai.py +++ b/tests/e2e/claude_code/thinking_with_tool_use/test_vertex_ai.py @@ -27,6 +27,8 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -75,6 +77,16 @@ def _has_block_type( @pytest.mark.covers("llm.messages.vertex.thinking_with_tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.REASONING, Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_thinking_with_tool_use_vertex_ai(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/tool_search/test_anthropic.py b/tests/e2e/claude_code/tool_search/test_anthropic.py index a23d1b3d2bb..f08c30b507e 100644 --- a/tests/e2e/claude_code/tool_search/test_anthropic.py +++ b/tests/e2e/claude_code/tool_search/test_anthropic.py @@ -45,6 +45,8 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_tool_search_shape, @@ -60,6 +62,16 @@ ANTHROPIC_MODELS = [ @pytest.mark.covers("llm.messages.anthropic.tool_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.TOOL_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_tool_search_anthropic(compat_result): """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` tool and assert the proxy + upstream accept it for every Anthropic diff --git a/tests/e2e/claude_code/tool_search/test_azure.py b/tests/e2e/claude_code/tool_search/test_azure.py index b094a35ea63..87628bc02d5 100644 --- a/tests/e2e/claude_code/tool_search/test_azure.py +++ b/tests/e2e/claude_code/tool_search/test_azure.py @@ -45,6 +45,8 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_tool_search_shape, @@ -61,6 +63,16 @@ AZURE_MODELS = [ @pytest.mark.skip(reason="stage red: Azure Foundry tool_search_server not supported in workspace for probed models") @pytest.mark.covers("llm.messages.azure_foundry.tool_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.TOOL_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_tool_search_azure(compat_result): """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` tool and assert the proxy + upstream accept it for every Azure (Microsoft Foundry) diff --git a/tests/e2e/claude_code/tool_search/test_bedrock_converse.py b/tests/e2e/claude_code/tool_search/test_bedrock_converse.py index f395122a5ab..84fd47bcb39 100644 --- a/tests/e2e/claude_code/tool_search/test_bedrock_converse.py +++ b/tests/e2e/claude_code/tool_search/test_bedrock_converse.py @@ -45,6 +45,8 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_tool_search_shape, @@ -60,6 +62,16 @@ BEDROCK_CONVERSE_MODELS = [ @pytest.mark.covers("llm.messages.bedrock_converse.tool_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.TOOL_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_tool_search_bedrock_converse(compat_result): """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` tool and assert the proxy + upstream accept it for every Bedrock (Converse) diff --git a/tests/e2e/claude_code/tool_search/test_bedrock_invoke.py b/tests/e2e/claude_code/tool_search/test_bedrock_invoke.py index 5b4c50e9dc5..b4a7f721a92 100644 --- a/tests/e2e/claude_code/tool_search/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/tool_search/test_bedrock_invoke.py @@ -50,6 +50,8 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_tool_search_replay_shape, @@ -67,6 +69,16 @@ BEDROCK_INVOKE_MODELS = [ @pytest.mark.covers("llm.messages.bedrock_invoke.tool_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.TOOL_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_tool_search_bedrock_invoke(compat_result): """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` tool and assert the proxy + upstream accept it for every Bedrock (Invoke) @@ -90,6 +102,16 @@ def test_tool_search_bedrock_invoke(compat_result): @pytest.mark.covers("llm.messages.bedrock_invoke.tool_search_history.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.TOOL_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_tool_search_history_bedrock_invoke(compat_result): """Send the tool-search request, take the real assistant turn back, and replay it as history with the tools still declared. diff --git a/tests/e2e/claude_code/tool_search/test_vertex_ai.py b/tests/e2e/claude_code/tool_search/test_vertex_ai.py index 7d0d35b1c1d..af629a7692e 100644 --- a/tests/e2e/claude_code/tool_search/test_vertex_ai.py +++ b/tests/e2e/claude_code/tool_search/test_vertex_ai.py @@ -45,6 +45,8 @@ from __future__ import annotations import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta + from claude_code._env import require_proxy_client from claude_code.http_probe import ( assert_tool_search_shape, @@ -61,6 +63,16 @@ VERTEX_AI_MODELS = [ @pytest.mark.skip(reason="stage red: Vertex rejects tool_search when deployment extra_headers inject context-1m beta; product/config") @pytest.mark.covers("llm.messages.vertex.tool_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.TOOL_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_tool_search_vertex_ai(compat_result): """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` tool and assert the proxy + upstream accept it for every Vertex AI diff --git a/tests/e2e/claude_code/tool_use/test_anthropic.py b/tests/e2e/claude_code/tool_use/test_anthropic.py index 9ff4c58907f..429557322f6 100644 --- a/tests/e2e/claude_code/tool_use/test_anthropic.py +++ b/tests/e2e/claude_code/tool_use/test_anthropic.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -72,6 +73,16 @@ def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.anthropic.tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_tool_use_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a tool call was emitted on the wire.""" diff --git a/tests/e2e/claude_code/tool_use/test_azure.py b/tests/e2e/claude_code/tool_use/test_azure.py index 9e7398267c4..96946093d9d 100644 --- a/tests/e2e/claude_code/tool_use/test_azure.py +++ b/tests/e2e/claude_code/tool_use/test_azure.py @@ -23,6 +23,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -66,6 +67,16 @@ def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.azure_foundry.tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_tool_use_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a tool call was emitted on the wire.""" diff --git a/tests/e2e/claude_code/tool_use/test_azure_openai.py b/tests/e2e/claude_code/tool_use/test_azure_openai.py index 7e1eecdbc03..3cb12b4023e 100644 --- a/tests/e2e/claude_code/tool_use/test_azure_openai.py +++ b/tests/e2e/claude_code/tool_use/test_azure_openai.py @@ -29,6 +29,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -67,6 +68,16 @@ def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool: return False +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE,), + models=tuple(AZURE_OPENAI_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_tool_use_azure_openai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a tool call was emitted on the wire by each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/tool_use/test_bedrock_converse.py b/tests/e2e/claude_code/tool_use/test_bedrock_converse.py index 33d4d3820d2..f361757042b 100644 --- a/tests/e2e/claude_code/tool_use/test_bedrock_converse.py +++ b/tests/e2e/claude_code/tool_use/test_bedrock_converse.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -62,6 +63,16 @@ def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.bedrock_converse.tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_tool_use_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a tool call was emitted on the wire.""" diff --git a/tests/e2e/claude_code/tool_use/test_bedrock_invoke.py b/tests/e2e/claude_code/tool_use/test_bedrock_invoke.py index 47ae3aef1da..16150c5eb9d 100644 --- a/tests/e2e/claude_code/tool_use/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/tool_use/test_bedrock_invoke.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -62,6 +63,16 @@ def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.bedrock_invoke.tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_tool_use_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a tool call was emitted on the wire.""" diff --git a/tests/e2e/claude_code/tool_use/test_bedrock_mantle.py b/tests/e2e/claude_code/tool_use/test_bedrock_mantle.py index e9cb70e74e9..b04b19e5542 100644 --- a/tests/e2e/claude_code/tool_use/test_bedrock_mantle.py +++ b/tests/e2e/claude_code/tool_use/test_bedrock_mantle.py @@ -33,6 +33,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code._gpt_cells import skip_unless_mantle_cells_enabled from claude_code.cli_driver import ( @@ -72,6 +73,16 @@ def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool: return False +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK_MANTLE,), + models=tuple(BEDROCK_MANTLE_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_tool_use_bedrock_mantle(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a tool call was emitted on the wire by each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/tool_use/test_openai.py b/tests/e2e/claude_code/tool_use/test_openai.py index ffb7e795c2b..d8a9ba83480 100644 --- a/tests/e2e/claude_code/tool_use/test_openai.py +++ b/tests/e2e/claude_code/tool_use/test_openai.py @@ -28,6 +28,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code._gpt_cells import skip_unless_openai_gpt_cells_enabled from claude_code.cli_driver import ( @@ -67,6 +68,16 @@ def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool: return False +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.OPENAI,), + models=tuple(OPENAI_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_tool_use_openai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a tool call was emitted on the wire by each GPT-5.6 tier.""" diff --git a/tests/e2e/claude_code/tool_use/test_vertex_ai.py b/tests/e2e/claude_code/tool_use/test_vertex_ai.py index 79a3016345c..a082a61920e 100644 --- a/tests/e2e/claude_code/tool_use/test_vertex_ai.py +++ b/tests/e2e/claude_code/tool_use/test_vertex_ai.py @@ -19,6 +19,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -62,6 +63,16 @@ def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.vertex.tool_use.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) +) def test_tool_use_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert a tool call was emitted on the wire.""" diff --git a/tests/e2e/claude_code/tool_use/test_vertex_ai_gpt.py b/tests/e2e/claude_code/tool_use/test_vertex_ai_gpt.py index d1ebbced9dc..c15ba00d325 100644 --- a/tests/e2e/claude_code/tool_use/test_vertex_ai_gpt.py +++ b/tests/e2e/claude_code/tool_use/test_vertex_ai_gpt.py @@ -20,9 +20,11 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations +from e2e_metadata import Domain, Subject, meta from claude_code._gpt_cells import VERTEX_AI_GPT_NOT_APPLICABLE_REASON +@meta(Subject(domain=Domain.LLM_TRANSLATION)) def test_tool_use_vertex_ai_gpt(compat_result): """Record the static not_applicable outcome for this cell.""" compat_result.set( diff --git a/tests/e2e/claude_code/tool_use_streaming/test_anthropic.py b/tests/e2e/claude_code/tool_use_streaming/test_anthropic.py index 152652dcf3c..91002e7ebf9 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_anthropic.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_anthropic.py @@ -29,6 +29,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -98,6 +99,16 @@ def _count_input_json_deltas(events: Sequence[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.anthropic.tool_use.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) +) def test_tool_use_streaming_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the proxy preserves fine-grained tool streaming end-to-end.""" diff --git a/tests/e2e/claude_code/tool_use_streaming/test_azure.py b/tests/e2e/claude_code/tool_use_streaming/test_azure.py index 8a1cc1852dd..4ec93f5a667 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_azure.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_azure.py @@ -21,6 +21,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -83,6 +84,16 @@ def _count_input_json_deltas(events: Sequence[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.azure_foundry.tool_use.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) +) def test_tool_use_streaming_azure(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/tool_use_streaming/test_azure_openai.py b/tests/e2e/claude_code/tool_use_streaming/test_azure_openai.py index ad5d4e0f613..020f4e79eb4 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_azure_openai.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_azure_openai.py @@ -31,6 +31,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -88,6 +89,16 @@ def _count_input_json_deltas(events: Sequence[Mapping[str, Any]]) -> int: ) +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE,), + models=tuple(AZURE_OPENAI_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) +) def test_tool_use_streaming_azure_openai(compat_result): proxy = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/tool_use_streaming/test_bedrock_converse.py b/tests/e2e/claude_code/tool_use_streaming/test_bedrock_converse.py index 3b04ed5962f..2d4b5764eb9 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_bedrock_converse.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_bedrock_converse.py @@ -27,6 +27,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -89,6 +90,16 @@ def _count_input_json_deltas(events: Sequence[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.bedrock_converse.tool_use.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) +) def test_tool_use_streaming_bedrock_converse(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/tool_use_streaming/test_bedrock_invoke.py b/tests/e2e/claude_code/tool_use_streaming/test_bedrock_invoke.py index c7b61129782..2cd7a68d57a 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_bedrock_invoke.py @@ -25,6 +25,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -87,6 +88,16 @@ def _count_input_json_deltas(events: Sequence[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.bedrock_invoke.tool_use.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) +) def test_tool_use_streaming_bedrock_invoke(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/tool_use_streaming/test_bedrock_mantle.py b/tests/e2e/claude_code/tool_use_streaming/test_bedrock_mantle.py index 20fae5d48db..190eb51adc0 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_bedrock_mantle.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_bedrock_mantle.py @@ -34,6 +34,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code._gpt_cells import skip_unless_mantle_cells_enabled from claude_code.cli_driver import ( @@ -92,6 +93,16 @@ def _count_input_json_deltas(events: Sequence[Mapping[str, Any]]) -> int: ) +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK_MANTLE,), + models=tuple(BEDROCK_MANTLE_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) +) def test_tool_use_streaming_bedrock_mantle(compat_result): skip_unless_mantle_cells_enabled() proxy = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/tool_use_streaming/test_openai.py b/tests/e2e/claude_code/tool_use_streaming/test_openai.py index a5ce31b1fd6..3555741b197 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_openai.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_openai.py @@ -29,6 +29,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code._gpt_cells import skip_unless_openai_gpt_cells_enabled from claude_code.cli_driver import ( @@ -87,6 +88,16 @@ def _count_input_json_deltas(events: Sequence[Mapping[str, Any]]) -> int: ) +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.OPENAI,), + models=tuple(OPENAI_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) +) def test_tool_use_streaming_openai(compat_result): skip_unless_openai_gpt_cells_enabled() proxy = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/tool_use_streaming/test_vertex_ai.py b/tests/e2e/claude_code/tool_use_streaming/test_vertex_ai.py index 2912e3aae3d..7bb29742cfd 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_vertex_ai.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_vertex_ai.py @@ -24,6 +24,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -86,6 +87,16 @@ def _count_input_json_deltas(events: Sequence[Mapping[str, Any]]) -> int: @pytest.mark.covers("llm.messages.vertex.tool_use.stream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.STREAM, + ) +) def test_tool_use_streaming_vertex_ai(compat_result): base_url, api_key = require_proxy(compat_result) diff --git a/tests/e2e/claude_code/tool_use_streaming/test_vertex_ai_gpt.py b/tests/e2e/claude_code/tool_use_streaming/test_vertex_ai_gpt.py index 7037e91fee0..5e6c203cb57 100644 --- a/tests/e2e/claude_code/tool_use_streaming/test_vertex_ai_gpt.py +++ b/tests/e2e/claude_code/tool_use_streaming/test_vertex_ai_gpt.py @@ -20,9 +20,11 @@ The (feature, provider) for this cell is inferred from the file path by from __future__ import annotations +from e2e_metadata import Domain, Subject, meta from claude_code._gpt_cells import VERTEX_AI_GPT_NOT_APPLICABLE_REASON +@meta(Subject(domain=Domain.LLM_TRANSLATION)) def test_tool_use_streaming_vertex_ai_gpt(compat_result): """Record the static not_applicable outcome for this cell.""" compat_result.set( diff --git a/tests/e2e/claude_code/vision/test_anthropic.py b/tests/e2e/claude_code/vision/test_anthropic.py index f681b2be5ae..b37bbb4ebf5 100644 --- a/tests/e2e/claude_code/vision/test_anthropic.py +++ b/tests/e2e/claude_code/vision/test_anthropic.py @@ -27,6 +27,7 @@ from __future__ import annotations import json import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -83,6 +84,16 @@ def _build_stdin_input() -> str: @pytest.mark.covers("llm.messages.anthropic.vision.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) +) def test_vision_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image attached via stream-json input and assert a non-empty reply.""" diff --git a/tests/e2e/claude_code/vision/test_azure.py b/tests/e2e/claude_code/vision/test_azure.py index f0eaaad84a2..739c8aba74d 100644 --- a/tests/e2e/claude_code/vision/test_azure.py +++ b/tests/e2e/claude_code/vision/test_azure.py @@ -27,6 +27,7 @@ from __future__ import annotations import json import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -83,6 +84,16 @@ def _build_stdin_input() -> str: @pytest.mark.covers("llm.messages.azure_foundry.vision.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) +) def test_vision_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image attached via stream-json input and assert a non-empty reply.""" diff --git a/tests/e2e/claude_code/vision/test_bedrock_converse.py b/tests/e2e/claude_code/vision/test_bedrock_converse.py index 2a5aba5a393..07f21b7d657 100644 --- a/tests/e2e/claude_code/vision/test_bedrock_converse.py +++ b/tests/e2e/claude_code/vision/test_bedrock_converse.py @@ -27,6 +27,7 @@ from __future__ import annotations import json import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -83,6 +84,16 @@ def _build_stdin_input() -> str: @pytest.mark.covers("llm.messages.bedrock_converse.vision.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) +) def test_vision_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image attached via stream-json input and assert a non-empty reply.""" diff --git a/tests/e2e/claude_code/vision/test_bedrock_invoke.py b/tests/e2e/claude_code/vision/test_bedrock_invoke.py index 5c995cd479e..847f2c80484 100644 --- a/tests/e2e/claude_code/vision/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/vision/test_bedrock_invoke.py @@ -27,6 +27,7 @@ from __future__ import annotations import json import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -83,6 +84,16 @@ def _build_stdin_input() -> str: @pytest.mark.covers("llm.messages.bedrock_invoke.vision.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) +) def test_vision_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image attached via stream-json input and assert a non-empty reply.""" diff --git a/tests/e2e/claude_code/vision/test_vertex_ai.py b/tests/e2e/claude_code/vision/test_vertex_ai.py index 8d385e295d0..256b762f06a 100644 --- a/tests/e2e/claude_code/vision/test_vertex_ai.py +++ b/tests/e2e/claude_code/vision/test_vertex_ai.py @@ -27,6 +27,7 @@ from __future__ import annotations import json import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -83,6 +84,16 @@ def _build_stdin_input() -> str: @pytest.mark.covers("llm.messages.vertex.vision.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.VISION,), + mode=Mode.NONSTREAM, + ) +) def test_vision_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image attached via stream-json input and assert a non-empty reply.""" diff --git a/tests/e2e/claude_code/web_search/test_anthropic.py b/tests/e2e/claude_code/web_search/test_anthropic.py index a20a2133dc9..8b7f4259a68 100644 --- a/tests/e2e/claude_code/web_search/test_anthropic.py +++ b/tests/e2e/claude_code/web_search/test_anthropic.py @@ -31,6 +31,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -85,6 +86,16 @@ def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.anthropic.web_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=tuple(ANTHROPIC_MODELS), + capabilities=(Capability.WEB_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_web_search_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream emitted a `tool_use` block calling `WebSearch`, proving diff --git a/tests/e2e/claude_code/web_search/test_azure.py b/tests/e2e/claude_code/web_search/test_azure.py index 8f9f638fbee..da3d1ebeb0b 100644 --- a/tests/e2e/claude_code/web_search/test_azure.py +++ b/tests/e2e/claude_code/web_search/test_azure.py @@ -31,6 +31,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -85,6 +86,16 @@ def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.azure_foundry.web_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.AZURE_AI,), + models=tuple(AZURE_MODELS), + capabilities=(Capability.WEB_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_web_search_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream emitted a `tool_use` block calling `WebSearch`, proving diff --git a/tests/e2e/claude_code/web_search/test_bedrock_converse.py b/tests/e2e/claude_code/web_search/test_bedrock_converse.py index 32f37b2be79..36c163dd2d6 100644 --- a/tests/e2e/claude_code/web_search/test_bedrock_converse.py +++ b/tests/e2e/claude_code/web_search/test_bedrock_converse.py @@ -31,6 +31,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -85,6 +86,16 @@ def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.bedrock_converse.web_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_CONVERSE_MODELS), + capabilities=(Capability.WEB_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_web_search_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream emitted a `tool_use` block calling `WebSearch`, proving diff --git a/tests/e2e/claude_code/web_search/test_bedrock_invoke.py b/tests/e2e/claude_code/web_search/test_bedrock_invoke.py index 68d1b30e83f..1328d361404 100644 --- a/tests/e2e/claude_code/web_search/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/web_search/test_bedrock_invoke.py @@ -31,6 +31,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -85,6 +86,16 @@ def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.bedrock_invoke.web_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=tuple(BEDROCK_INVOKE_MODELS), + capabilities=(Capability.WEB_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_web_search_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream emitted a `tool_use` block calling `WebSearch`, proving diff --git a/tests/e2e/claude_code/web_search/test_vertex_ai.py b/tests/e2e/claude_code/web_search/test_vertex_ai.py index 540a8396c98..7ab352b1953 100644 --- a/tests/e2e/claude_code/web_search/test_vertex_ai.py +++ b/tests/e2e/claude_code/web_search/test_vertex_ai.py @@ -31,6 +31,7 @@ from typing import Any, Mapping, Sequence import pytest +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, @@ -85,6 +86,16 @@ def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: @pytest.mark.covers("llm.messages.vertex.web_search.nonstream.works") +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.VERTEX_AI,), + models=tuple(VERTEX_AI_MODELS), + capabilities=(Capability.WEB_SEARCH,), + mode=Mode.NONSTREAM, + ) +) def test_web_search_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the upstream emitted a `tool_use` block calling `WebSearch`, proving From d7cdc88c66a89db4a837c26d5d0719c7b40823ba Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Wed, 7 Oct 2026 10:32:09 -0700 Subject: [PATCH 04/13] test(e2e): tag management tests with Subject metadata and record management client steps (#44962) * test(e2e): add enum values, auto-discovering label gates and secret hiding for e2e metadata * test(e2e): tag management tests with Subject metadata and record management client steps * docs(e2e): name every markerless harness test file that carries no Subject * test(e2e): keep the step discovery comprehensions to one for clause * test(e2e): keep the prompt out of the chat_status step so polled retries collapse --- tests/e2e/management/jwt_actors.py | 5 + tests/e2e/management/management_client.py | 68 +++- .../test_budget_customer_user_org_e2e.py | 14 + .../test_config_misc_endpoints_e2e.py | 12 + .../e2e/management/test_jwt_management_e2e.py | 10 + .../e2e/management/test_key_lifecycle_e2e.py | 22 ++ .../e2e/management/test_key_management_e2e.py | 29 +- tests/e2e/management/test_management_e2e.py | 308 +++++++++++++++--- .../e2e/management/test_mcp_lifecycle_e2e.py | 61 ++++ .../test_model_tag_accessgroup_e2e.py | 37 +++ .../test_model_test_connection_e2e.py | 10 + .../management/test_team_management_e2e.py | 15 + 12 files changed, 545 insertions(+), 46 deletions(-) diff --git a/tests/e2e/management/jwt_actors.py b/tests/e2e/management/jwt_actors.py index 2d23549fe71..ae33f583195 100644 --- a/tests/e2e/management/jwt_actors.py +++ b/tests/e2e/management/jwt_actors.py @@ -5,6 +5,7 @@ from typing import Final, Literal from e2e_config import unique_marker from e2e_http import NoBody, unwrap +from e2e_metadata import step from idp import ADMIN_CLIENT_ID, TESTS_CLIENT_ID, Identity, Keycloak from lifecycle import ResourceManager from management.management_client import ManagementClient @@ -53,6 +54,7 @@ class Actor: profile: ActorProfile tenants: tuple[Tenant, ...] + @step("Get a JWT from the identity provider for the actor in the {self.role} role") def mint_caller(self, idp: Keycloak) -> Caller: return Caller( credential=idp.access_token( @@ -74,6 +76,7 @@ class ActorFactory: if self.bootstrap.proxy.caller is not None: raise ValueError("Actor bootstrap requires a separately held master client") + @step("Generate a virtual key as the proxy admin") def key(self, tenant: Tenant | None = None, *, user_id: str | None = None) -> KeyGenerateResponse: created: Final = unwrap( self.bootstrap.generate_key( @@ -87,6 +90,7 @@ class ActorFactory: self.resources.defer(lambda: self.bootstrap.delete_key_strict(created.key, missing_ok=True)) return created + @step("Create an organization, a team in it and a matching identity provider group") def tenant(self) -> Tenant: marker: Final = unique_marker() organization_id: Final = self.bootstrap.create_org(OrgNewBody(organization_alias=f"e2e-organization-{marker}")) @@ -118,6 +122,7 @@ class ActorFactory: self.resources.defer(lambda: self.idp.with_strict_cleanup().delete_group(group_id)) return Tenant(organization_id=organization_id, team_id=team_id, group_id=group_id) + @step("Create an actor in the {role} role, with its identity provider user, internal user and any tenant memberships") def create( self, role: ActorRole, *, tenants: tuple[Tenant, ...] = (), profile: ActorProfile = "database_role" ) -> Actor: diff --git a/tests/e2e/management/management_client.py b/tests/e2e/management/management_client.py index 3da9bea12a3..ffa310a9f4a 100644 --- a/tests/e2e/management/management_client.py +++ b/tests/e2e/management/management_client.py @@ -24,6 +24,7 @@ from e2e_http import ( retry_attempts, unwrap, ) +from e2e_metadata import STEP_FRAMES, step from models import ( AuditLogPage, AuditLogParams, @@ -116,9 +117,11 @@ class ManagementClient: def with_caller(self, caller: Caller) -> ManagementClient: return replace(self, proxy=self.proxy.with_caller(caller)) + @step("Generate a virtual key limited to the LLM API routes") def llm_only_key(self) -> str: return self.proxy.generate_key(KeyGenerateBody(models=[], allowed_routes=["llm_api_routes"])) + @step("Generate a virtual key with {body}") def generate_key(self, body: KeyGenerateBody, *, caller_key: str | None = None) -> Result[KeyGenerateResponse]: """POST /key/generate. `caller_key` is who is creating the key: the master key by default, or a virtual key (an admin filling in Create New Key on the @@ -133,6 +136,7 @@ class ManagementClient: response_type=KeyGenerateResponse, ) + @step("Update the virtual key's settings with /key/update") def update_key(self, body: KeyUpdateBody, *, caller_key: str | None = None) -> Result[NoBody]: """POST /key/update. `caller_key` is who is editing: the master key by default, or a virtual key (the dashboard edits under the session key its @@ -152,16 +156,22 @@ class ManagementClient: case UnknownApiError(body=error_body) if any( marker in error_body.lower() for marker in _TRANSIENT_BACKEND_MARKERS ): - warnings.warn(f"Transient backend response on attempt {attempt + 1}", RuntimeWarning, stacklevel=2) + warnings.warn( + f"Transient backend response on attempt {attempt + 1}", + RuntimeWarning, + stacklevel=2 + STEP_FRAMES, + ) time.sleep(0.5 * (attempt + 1)) continue case _: break return last + @step("Set the virtual key's models to [{models}]") def update_key_models(self, key: str, models: list[str]) -> None: _ = unwrap(self.update_key(KeyUpdateBody(key=key, models=models))) + @step("Delete the virtual key with the alias {key_alias}") def delete_key_by_alias(self, key_alias: str) -> None: _ = unwrap( self.proxy.transport.post( @@ -172,6 +182,7 @@ class ManagementClient: ) ) + @step("Read the key's deletion entries from the /audit log") def key_deleted_audit_logs(self, token_hash: str) -> AuditLogPage: return unwrap( self.proxy.transport.get( @@ -187,6 +198,7 @@ class ManagementClient: ) ) + @step("Read the key's settings back from /key/info") def key_info_as(self, key: str, *, caller_key: str | None = None) -> Result[KeyInfoResponse]: return self.proxy.transport.get( "/key/info", @@ -195,6 +207,7 @@ class ManagementClient: response_type=KeyInfoResponse, ) + @step("Delete the virtual key") def delete_key_strict(self, key: str, *, caller_key: str | None = None, missing_ok: bool = False) -> None: """Strict delete for the act phase of a test: a failed delete is a hard failure, unlike the warn-only ProxyClient.delete_key used at teardown.""" @@ -208,6 +221,7 @@ class ManagementClient: return _ = unwrap(result) + @step("Delete the deployment") def delete_model_strict(self, model_id: str) -> None: """Strict delete for the act phase of a test: a failed delete is a hard failure, unlike the warn-only ProxyClient.delete_model used at teardown.""" @@ -220,6 +234,7 @@ class ManagementClient: ) ) + @step("Run Test Connection on {body.litellm_params.model} in {body.mode} mode with /health/test_connection") def connection_test(self, body: ConnectionTestBody) -> Result[ConnectionTestResponse]: """POST /health/test_connection, the call behind the Admin UI's Test Connection button, probing the live provider with the supplied params.""" @@ -231,6 +246,7 @@ class ManagementClient: timeout=120.0, ) + @step("Block the virtual key") def block_key(self, key: str) -> None: _ = unwrap( self.proxy.transport.post( @@ -240,6 +256,7 @@ class ManagementClient: response_type=NoBody, ) ) + @step("Regenerate the virtual key with /key/regenerate") def regenerate_key(self, key: str, *, grace_period: str | None = None) -> str: return unwrap( self.proxy.transport.post( @@ -250,6 +267,7 @@ class ManagementClient: ) ).key + @step("Reset the virtual key's spend to {reset_to}") def reset_key_spend(self, key: str, reset_to: float) -> KeyResetSpendResponse: return unwrap( self.proxy.transport.post( @@ -260,6 +278,7 @@ class ManagementClient: ) ) + @step("List the keys with the alias {key_alias} from /key/list") def key_list(self, key_alias: str, *, caller_key: str | None = None) -> Result[KeyListResponse]: """GET /key/list, the Virtual Keys page's own inventory call. `caller_key` is who is asking: the master key by default, or a virtual key.""" @@ -271,9 +290,11 @@ class ManagementClient: response_type=KeyListResponse, ) + @step("Count the keys with the alias {key_alias} in /key/list") def key_alias_count(self, key_alias: str) -> int: return unwrap(self.key_list(key_alias)).total_count + @step("Sign in to the Admin UI with /v2/login") def dashboard_login(self, username: str, password: str) -> DashboardSession: """POST /v2/login, the call the Admin UI's sign-in form makes. @@ -297,6 +318,7 @@ class ManagementClient: redirect_url=response.redirect_url, ) + @step("Create a team with {body}") def create_team(self, body: TeamNewBody) -> str: team_id = unwrap( self.proxy.transport.post( @@ -309,6 +331,7 @@ class ManagementClient: self._wait_for_team(team_id) return team_id + @step("Update a team with {body}") def update_team(self, body: TeamUpdateBody) -> None: last: Result[NoBody] | None = None for attempt in range(retry_attempts(5)): @@ -324,7 +347,11 @@ class ManagementClient: case UnknownApiError(body=body_text) if ( "connecting to redis" in body_text.lower() or "name resolution" in body_text.lower() ): - warnings.warn(f"Transient backend response on attempt {attempt + 1}", RuntimeWarning, stacklevel=2) + warnings.warn( + f"Transient backend response on attempt {attempt + 1}", + RuntimeWarning, + stacklevel=2 + STEP_FRAMES, + ) time.sleep(0.5 * (attempt + 1)) continue case _: @@ -332,6 +359,7 @@ class ManagementClient: assert last is not None raise AssertionError(last) + @step("Delete the team") def delete_team(self, team_id: str) -> None: _ = self.proxy.transport.post( "/team/delete", @@ -340,6 +368,7 @@ class ManagementClient: response_type=NoBody, ) + @step("Read the team back from /team/info") def team_info(self, team_id: str) -> TeamData: return unwrap( self.proxy.transport.get( @@ -350,6 +379,7 @@ class ManagementClient: ) ).team_info + @step("List the teams from /team/list") def team_list_ids(self) -> tuple[str, ...]: return tuple( entry.team_id @@ -363,6 +393,7 @@ class ManagementClient: ).root ) + @step("Check whether /team/info finds the team") def team_info_status(self, team_id: str) -> ProbeResult: return self.proxy.transport.probe( "/team/info", params=TeamInfoParams(team_id=team_id), headers=self.proxy.management_headers() @@ -386,6 +417,7 @@ class ManagementClient: assert last is not None raise AssertionError(last) + @step("Add a user to the team with /team/member_add") def add_team_member(self, team_id: str, user_id: str) -> None: last: Result[NoBody] | None = None for attempt in range(retry_attempts(_TEAM_READY_ATTEMPTS)): @@ -402,7 +434,9 @@ class ManagementClient: _TEAM_READY_ATTEMPTS ): warnings.warn( - "Retrying team membership while the team becomes available", RuntimeWarning, stacklevel=2 + "Retrying team membership while the team becomes available", + RuntimeWarning, + stacklevel=2 + STEP_FRAMES, ) time.sleep(_TEAM_READY_SLEEP_SECONDS) continue @@ -411,6 +445,7 @@ class ManagementClient: assert last is not None raise AssertionError(last) + @step("Add a roster of members to the team with /team/member_add") def add_team_members(self, team_id: str, members: list[TeamMemberEntry]) -> None: """Bulk form of /team/member_add: `member` accepts a list, so one call seeds a whole roster the way an admin import does.""" @@ -423,6 +458,7 @@ class ManagementClient: ) ) + @step("Try to delete the team with /team/delete") def delete_team_status(self, team_id: str) -> StreamingResponse: """POST /team/delete judged by HTTP outcome: the raw status and body, so a test can assert on what a caller actually sees when the delete fails.""" @@ -432,6 +468,7 @@ class ManagementClient: json=TeamDeleteBody(team_ids=[team_id]), ) + @step("Remove a user from the team with /team/member_delete") def delete_team_member(self, team_id: str, user_id: str) -> None: _ = unwrap( self.proxy.transport.post( @@ -442,6 +479,7 @@ class ManagementClient: ) ) + @step("Create an internal user with {body}") def create_user(self, body: UserNewBody) -> str: return unwrap( self.proxy.transport.post( @@ -452,6 +490,7 @@ class ManagementClient: ) ).user_id + @step("Create the end user {user_id}") def create_customer(self, user_id: str) -> str: _ = unwrap( self.proxy.transport.post( @@ -463,6 +502,7 @@ class ManagementClient: ) return user_id + @step("Read the end user {end_user_id} back from /customer/info") def customer_info(self, end_user_id: str) -> CustomerResponse: return unwrap( self.proxy.transport.get( @@ -473,6 +513,7 @@ class ManagementClient: ) ) + @step("Delete the end user {user_id}") def delete_customer(self, user_id: str) -> None: _ = self.proxy.transport.post( "/customer/delete", @@ -481,6 +522,7 @@ class ManagementClient: response_type=NoBody, ) + @step("Update an internal user with {body}") def update_user(self, body: UserUpdateBody) -> None: _ = unwrap( self.proxy.transport.post( @@ -491,6 +533,7 @@ class ManagementClient: ) ) + @step("Delete the internal user") def delete_user(self, user_id: str) -> None: _ = self.proxy.transport.post( "/user/delete", @@ -499,6 +542,7 @@ class ManagementClient: response_type=NoBody, ) + @step("Delete the internal user") def delete_user_strict(self, user_id: str) -> None: """Strict delete for the act phase of a test: a failed delete is a hard failure, unlike the warn-only delete_user used at teardown.""" @@ -511,6 +555,7 @@ class ManagementClient: ) ) + @step("Read the user back from /user/info") def user_info(self, user_id: str | None = None) -> UserInfoResponse: return unwrap( self.proxy.transport.get( @@ -521,6 +566,7 @@ class ManagementClient: ) ) + @step("Count the matching users in /user/list") def user_count(self, user_id: str) -> int: return unwrap( self.proxy.transport.get( @@ -531,6 +577,7 @@ class ManagementClient: ) ).total + @step("List the matching users from /user/list") def user_list_ids(self, user_id: str) -> tuple[str, ...]: listing = unwrap( self.proxy.transport.get( @@ -542,6 +589,7 @@ class ManagementClient: ) return tuple(row.user_id for row in listing.users) + @step("Create an organization with {body}") def create_org(self, body: OrgNewBody) -> str: return unwrap( self.proxy.transport.post( @@ -552,6 +600,7 @@ class ManagementClient: ) ).organization_id + @step("Update an organization with {body}") def update_org(self, body: OrgUpdateBody) -> None: _ = unwrap( self.proxy.transport.patch( @@ -562,6 +611,7 @@ class ManagementClient: ) ) + @step("Delete the organization") def delete_org(self, organization_id: str) -> None: _ = self.proxy.transport.delete( "/organization/delete", @@ -570,6 +620,7 @@ class ManagementClient: response_type=NoBody, ) + @step("Read the organization back from /organization/info") def org_info(self, organization_id: str) -> OrgInfoResponse: return unwrap( self.proxy.transport.get( @@ -580,6 +631,7 @@ class ManagementClient: ) ) + @step("Check whether /organization/info finds the organization") def org_info_status(self, organization_id: str) -> ProbeResult: return self.proxy.transport.probe( "/organization/info", @@ -587,6 +639,7 @@ class ManagementClient: headers=self.proxy.management_headers(), ) + @step("Create a tag with {body}") def create_tag(self, body: TagNewBody) -> None: _ = unwrap( self.proxy.transport.post( @@ -597,6 +650,7 @@ class ManagementClient: ) ) + @step("Delete the tag {name}") def delete_tag(self, name: str) -> None: _ = self.proxy.transport.post( "/tag/delete", @@ -605,6 +659,7 @@ class ManagementClient: response_type=NoBody, ) + @step("List the tags from /tag/list") def tag_list(self) -> tuple[TagListEntry, ...]: return tuple( unwrap( @@ -617,6 +672,7 @@ class ManagementClient: ).root ) + @step("Create an MCP server named {body.alias}") def create_mcp_server(self, body: McpServerCreateBody) -> McpServerRow: return unwrap( self.proxy.transport.post( @@ -627,6 +683,7 @@ class ManagementClient: ) ) + @step("Update the MCP server's settings with PUT /v1/mcp/server") def update_mcp_server(self, body: McpServerUpdateBody) -> McpServerRow: """PUT /v1/mcp/server, the call behind the dashboard's Save Changes: a partial update where a field left unset keeps its stored value and None clears it.""" @@ -639,6 +696,7 @@ class ManagementClient: ) ) + @step("Delete the MCP server") def delete_mcp_server(self, server_id: str) -> Result[NoBody]: """DELETE /v1/mcp/server/{server_id}. Returns the outcome so the act phase can unwrap it while a deferred teardown can ignore an already-deleted server.""" @@ -649,6 +707,7 @@ class ManagementClient: response_type=NoBody, ) + @step("Send a /chat/completions request to {model}") def chat_status(self, key: str, model: str, content: str) -> StreamingResponse: return self.proxy.transport.send( "/chat/completions", @@ -656,12 +715,15 @@ class ManagementClient: json=ChatBody(model=model, messages=[ChatMessage(role="user", content=content)], max_tokens=16), ) + @step("Try to generate a virtual key with {body}, calling with a virtual key") def key_generate_status(self, key: str, body: KeyGenerateBody) -> StreamingResponse: return self.proxy.transport.send("/key/generate", headers=self.proxy.transport.bearer(key), json=body) + @step("Try to create a team with {body}, calling with a virtual key") def team_new_status(self, key: str, body: TeamNewBody) -> StreamingResponse: return self.proxy.transport.send("/team/new", headers=self.proxy.transport.bearer(key), json=body) + @step("Try to create an internal user with {body}, calling with a virtual key") def user_new_status(self, key: str, body: UserNewBody) -> StreamingResponse: return self.proxy.transport.send("/user/new", headers=self.proxy.transport.bearer(key), json=body) diff --git a/tests/e2e/management/test_budget_customer_user_org_e2e.py b/tests/e2e/management/test_budget_customer_user_org_e2e.py index 6e14a2d5745..ae4e6db4e15 100644 --- a/tests/e2e/management/test_budget_customer_user_org_e2e.py +++ b/tests/e2e/management/test_budget_customer_user_org_e2e.py @@ -23,6 +23,7 @@ from pydantic import BaseModel, Field, RootModel from e2e_config import unique_marker from e2e_http import NoBody, Success, UnauthorizedError, UnknownApiError, is_ok, unwrap +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager from management_client import ManagementClient from models import KeyGenerateBody, ModelBudgetEntry, OrgInfoParams, OrgNewBody, UserNewBody @@ -152,6 +153,7 @@ _UPDATED_MAX_BUDGET = 91.25 class TestBudgetManagement: @pytest.mark.covers("mgmt.budget.list.happy_path") + @meta(Subject(domain=Domain.SPEND_BUDGETS, route=Route.BUDGET_MANAGEMENT)) def test_created_budget_appears_in_budget_list( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -164,6 +166,7 @@ class TestBudgetManagement: ) @pytest.mark.covers("mgmt.budget.update.accepts_model_max_budget") + @meta(Subject(domain=Domain.SPEND_BUDGETS, route=Route.BUDGET_MANAGEMENT)) def test_update_accepts_per_model_budgets_including_punctuated_names( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -213,6 +216,7 @@ class TestBudgetManagement: ) @pytest.mark.covers("mgmt.budget.update.persists") + @meta(Subject(domain=Domain.SPEND_BUDGETS, route=Route.BUDGET_MANAGEMENT)) def test_update_max_budget_persists_to_budget_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -247,6 +251,7 @@ class TestBudgetManagement: ) @pytest.mark.covers("mgmt.budget.new.admin_only") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.BUDGET_MANAGEMENT)) def test_new_is_refused_for_a_non_admin_key( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -352,6 +357,7 @@ class TestBudgetListV1: """ @pytest.mark.covers("mgmt.budget.list_v1.happy_path") + @meta(Subject(domain=Domain.SPEND_BUDGETS, route=Route.BUDGET_MANAGEMENT)) def test_sorts_pages_and_filters_the_budgets_it_created( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -392,6 +398,7 @@ class TestBudgetListV1: assert [row.tpm_limit for row in limits] == [60000, 60000, 60000] @pytest.mark.covers("mgmt.budget.list_v1.happy_path") + @meta(Subject(domain=Domain.SPEND_BUDGETS, route=Route.BUDGET_MANAGEMENT)) def test_is_null_finds_the_budget_left_uncapped( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -409,11 +416,13 @@ class TestBudgetListV1: assert [row.max_budget for row in _list_budgets(client, found).data] == [None] @pytest.mark.covers("mgmt.budget.list_v1.happy_path") + @meta(Subject(domain=Domain.SPEND_BUDGETS, route=Route.BUDGET_MANAGEMENT)) def test_refuses_a_sort_field_and_a_parameter_it_does_not_support(self, client: ManagementClient) -> None: assert _list_status(client, BudgetPageParams(sort="budget_duration")) == 400 assert _list_status(client, BudgetPageParams(not_a_parameter="b-1")) == 400 @pytest.mark.covers("mgmt.budget.list_v1.admin_only") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.BUDGET_MANAGEMENT)) def test_is_refused_for_a_non_admin_key(self, client: ManagementClient, resources: ResourceManager) -> None: key = client.proxy.generate_key(KeyGenerateBody()) resources.defer(lambda: client.proxy.delete_key(key)) @@ -482,6 +491,7 @@ def _customer_info(client: ManagementClient, route: str, user_id: str) -> Custom class TestCustomerManagement: @pytest.mark.covers("mgmt.customer.new.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.CUSTOMER_MANAGEMENT)) def test_new_persists_to_customer_info(self, client: ManagementClient, resources: ResourceManager) -> None: customer_id = f"e2e-mgmt-cust-{unique_marker()}" created = _create_customer( @@ -495,6 +505,7 @@ class TestCustomerManagement: ) @pytest.mark.covers("mgmt.customer.delete.persists") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.CUSTOMER_MANAGEMENT)) def test_delete_removes_the_customer(self, client: ManagementClient, resources: ResourceManager) -> None: """The teardown's deferred delete fires again on the already-deleted customer by design: it is the safety net if this test fails before the in-body delete, @@ -524,6 +535,7 @@ class TestCustomerManagement: _ = _poll(client, gone, f"customer {customer_id} still resolved on /customer/info after /customer/delete") @pytest.mark.covers("mgmt.end_user.new.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.CUSTOMER_MANAGEMENT)) def test_end_user_new_persists_to_end_user_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -542,6 +554,7 @@ class TestCustomerManagement: class TestUserManagement: @pytest.mark.covers("mgmt.user.info.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.USER_MANAGEMENT)) def test_new_user_is_readable_via_user_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -586,6 +599,7 @@ class OrgInfoMembersResponse(BaseModel): class TestOrganizationMembership: @pytest.mark.covers("mgmt.organization.member_add.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.ORGANIZATION_MANAGEMENT)) def test_member_add_records_membership( self, client: ManagementClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/management/test_config_misc_endpoints_e2e.py b/tests/e2e/management/test_config_misc_endpoints_e2e.py index a3be0a64e7f..b7ab311cc18 100644 --- a/tests/e2e/management/test_config_misc_endpoints_e2e.py +++ b/tests/e2e/management/test_config_misc_endpoints_e2e.py @@ -32,6 +32,7 @@ from pydantic import BaseModel from e2e_config import unique_marker from e2e_http import NoBody, Success, unwrap, unwrap_status +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager from management_client import ManagementClient from models import KeyGenerateBody, LiteLLMParamsBody, TeamNewBody @@ -231,6 +232,7 @@ class McpServerResponse(BaseModel): class TestInventoryRoutes: @pytest.mark.covers("mgmt.callback.list.happy_path") + @meta(Subject(domain=Domain.OBSERVABILITY)) def test_callbacks_list_reports_active_logging_callbacks(self, client: ManagementClient) -> None: listing = unwrap( client.proxy.transport.get( @@ -247,6 +249,7 @@ class TestInventoryRoutes: ) @pytest.mark.covers("mgmt.tool_management.list.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT)) def test_tool_list_returns_catalog_with_consistent_total(self, client: ManagementClient) -> None: listing = unwrap( client.proxy.transport.get( @@ -261,6 +264,7 @@ class TestInventoryRoutes: ) @pytest.mark.covers("mgmt.workflow.list.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT)) def test_workflow_runs_list_returns_consistent_count(self, client: ManagementClient) -> None: listing = unwrap( client.proxy.transport.get( @@ -275,6 +279,7 @@ class TestInventoryRoutes: ) @pytest.mark.covers("mgmt.credential_migration.check.happy_path") + @meta(Subject(domain=Domain.DEPLOY_OPS)) def test_credential_migration_check_reports_residual_scan(self, client: ManagementClient) -> None: report = unwrap( client.proxy.transport.get( @@ -295,6 +300,7 @@ class TestInventoryRoutes: class TestCostEstimate: @pytest.mark.covers("mgmt.cost_tracking.estimate.happy_path") + @meta(Subject(domain=Domain.COST_MAP)) def test_estimate_computes_cost_from_token_counts(self, client: ManagementClient) -> None: estimate = unwrap( client.proxy.transport.post( @@ -325,6 +331,7 @@ class TestCostEstimate: class TestComplianceRoutes: @pytest.mark.covers("mgmt.compliance.gdpr.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT)) def test_gdpr_check_derives_verdict_from_the_request(self, client: ManagementClient) -> None: result = unwrap( client.proxy.transport.post( @@ -356,6 +363,7 @@ class TestComplianceRoutes: class TestFallbackManagement: @pytest.mark.covers("mgmt.fallback_management.update.happy_path") + @meta(Subject(domain=Domain.ROUTING)) def test_create_persists_and_is_read_back(self, client: ManagementClient, resources: ResourceManager) -> None: primary = f"e2e-fallback-primary-{unique_marker()}" secondary = f"e2e-fallback-secondary-{unique_marker()}" @@ -410,6 +418,7 @@ class TestFallbackManagement: class TestJwtKeyMapping: @pytest.mark.covers("mgmt.jwt_key_mapping.new.happy_path") + @meta(Subject(domain=Domain.PROXY_AUTH)) def test_new_persists_mapping_and_is_read_back( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -461,6 +470,7 @@ class TestJwtKeyMapping: class TestRouterSettings: @pytest.mark.covers("mgmt.router_settings.update.happy_path") + @meta(Subject(domain=Domain.ROUTING, route=Route.PROXY_CONFIG)) def test_config_update_persists_router_setting_to_get( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -531,6 +541,7 @@ class TestRouterSettings: class TestMcpServerSubmission: @pytest.mark.covers("mgmt.mcp_server.register.happy_path") + @meta(Subject(domain=Domain.MCP, route=Route.MCP)) def test_register_submits_pending_server(self, client: ManagementClient, resources: ResourceManager) -> None: """A non-admin, team-scoped key submits an MCP server for review; the proxy stores it as pending_review without loading it into the runtime registry.""" @@ -564,6 +575,7 @@ class TestMcpServerSubmission: ) @pytest.mark.covers("mgmt.mcp_server.approve.persists") + @meta(Subject(domain=Domain.MCP, route=Route.MCP)) def test_approve_activates_submission_and_persists( self, client: ManagementClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/management/test_jwt_management_e2e.py b/tests/e2e/management/test_jwt_management_e2e.py index 5898073a4e6..3853ae2c4e6 100644 --- a/tests/e2e/management/test_jwt_management_e2e.py +++ b/tests/e2e/management/test_jwt_management_e2e.py @@ -7,6 +7,7 @@ from typing import Final, Literal import pytest from e2e_config import CHEAP_OPENAI_MODEL, PROXY_BASE_URL, unique_marker from e2e_http import UnauthorizedError, UnknownApiError, unwrap +from e2e_metadata import Domain, Route, Subject, meta from idp import ADMIN_CLIENT_ID, Identity, Keycloak, token_claims from lifecycle import ResourceManager from management.jwt_actors import ActorFactory, ActorRole @@ -32,6 +33,7 @@ class TestJwtManagement: ), ) @pytest.mark.covers("mgmt.user.jwt.database_roles") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.USER_MANAGEMENT)) def test_actor_subject_and_database_role(self, actor_factory: ActorFactory, role: ActorRole) -> None: tenants: Final = ( (actor_factory.tenant(),) if role in ("organization_admin", "team_admin", "team_member") else () @@ -70,6 +72,7 @@ class TestJwtManagement: } == {(actor.identity.user_id, "org_admin" if role == "organization_admin" else "internal_user")} @pytest.mark.covers("mgmt.key.jwt.viewer_denied") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.KEY_MANAGEMENT)) def test_admin_viewer_reads_but_cannot_update(self, actor_factory: ActorFactory) -> None: actor: Final = actor_factory.create("proxy_admin_viewer") viewer: Final = actor_factory.bootstrap.with_caller(actor.mint_caller(actor_factory.idp)) @@ -83,6 +86,7 @@ class TestJwtManagement: assert actor_factory.bootstrap.proxy.key_info(key).key_alias == alias @pytest.mark.covers("mgmt.user.oidc.identity_mapping") + @meta(Subject(domain=Domain.PROXY_AUTH)) def test_oidc_browser_profile_identity_mapping(self, actor_factory: ActorFactory) -> None: actor: Final = actor_factory.create("internal_user") idp: Final = actor_factory.idp.with_strict_cleanup() @@ -100,6 +104,7 @@ class TestJwtManagement: @pytest.mark.covers("mgmt.key.jwt.lifecycle") @pytest.mark.parametrize("credential_kind", ("direct_jwt", "virtual_key")) + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.KEY_MANAGEMENT)) def test_admin_creates_reads_updates_clears_and_deletes_a_key( self, actor_factory: ActorFactory, @@ -142,6 +147,7 @@ class TestJwtManagement: assert unwrap(bound.key_list(updated_alias)).total_count == 0 @pytest.mark.covers("mgmt.team.jwt.tenant_isolation") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.KEY_MANAGEMENT)) def test_two_actor_sets_keep_tenants_and_keys_isolated(self, actor_factory: ActorFactory) -> None: first: Final = actor_factory.tenant() second: Final = actor_factory.tenant() @@ -163,6 +169,7 @@ class TestJwtManagement: assert tuple(actor.identity.groups for actor in actors) == ((first.team_id,), (second.team_id,)) @pytest.mark.covers("mgmt.team.jwt.multiple_memberships") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_multi_group_actor_keeps_exact_memberships(self, actor_factory: ActorFactory) -> None: tenants: Final = (actor_factory.tenant(), actor_factory.tenant()) actor: Final = actor_factory.create("team_member", tenants=tenants, profile="group_scoped") @@ -177,6 +184,7 @@ class TestJwtManagement: } == {(actor.identity.user_id, "user")} @pytest.mark.covers("mgmt.user.jwt.cleanup") + @meta(Subject(domain=Domain.PROXY_AUTH)) def test_successful_actor_cleanup_removes_owned_state(self, actor_factory: ActorFactory) -> None: resources: Final = ResourceManager(client=actor_factory.bootstrap.proxy, strict_cleanup=True) factory: Final = ActorFactory(bootstrap=actor_factory.bootstrap, idp=actor_factory.idp, resources=resources) @@ -197,6 +205,7 @@ class TestJwtManagement: @pytest.mark.parametrize("stage", ("group", "user")) @pytest.mark.covers("mgmt.user.jwt.partial_cleanup") + @meta(Subject(domain=Domain.PROXY_AUTH)) def test_partial_setup_removes_previously_created_identities( self, actor_factory: ActorFactory, @@ -236,6 +245,7 @@ class TestJwtManagement: idp.assert_absent("users", identity.user_id) @pytest.mark.covers("mgmt.key.jwt.member_denied", "mgmt.key.jwt.other_team_denied") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.KEY_MANAGEMENT)) def test_member_cannot_write_and_another_team_cannot_read_the_key( self, client: ManagementClient, idp: Keycloak, jwt_identity: Identity, resources: ResourceManager ) -> None: diff --git a/tests/e2e/management/test_key_lifecycle_e2e.py b/tests/e2e/management/test_key_lifecycle_e2e.py index fb153f2a7f3..956719e0ba3 100644 --- a/tests/e2e/management/test_key_lifecycle_e2e.py +++ b/tests/e2e/management/test_key_lifecycle_e2e.py @@ -23,6 +23,7 @@ import pytest from e2e_config import unique_marker from e2e_http import Result, StreamingResponse, Success, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from management_client import MODEL_ACCESS_DENIED_MARKER, ManagementClient from models import ( @@ -188,6 +189,7 @@ def _assert_chat_rejected_everywhere(client: ManagementClient, key: str, model: class TestKeyLifecycle: + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.KEY_MANAGEMENT)) def test_create_echoes_every_field_written( self, client: ManagementClient, resources: ResourceManager, mock_deployment: str ) -> None: @@ -206,6 +208,7 @@ class TestKeyLifecycle: ): assert observed == wanted, f"/key/generate echoed {field}={observed!r}, sent {wanted!r}" + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.KEY_MANAGEMENT)) def test_read_reflects_the_create_on_every_replica( self, client: ManagementClient, resources: ResourceManager, mock_deployment: str ) -> None: @@ -221,6 +224,7 @@ class TestKeyLifecycle: ) @pytest.mark.covers("mgmt.key.update.preserves_unrelated_fields") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.KEY_MANAGEMENT)) def test_partial_update_changes_only_the_named_field( self, client: ManagementClient, resources: ResourceManager, mock_deployment: str ) -> None: @@ -238,6 +242,7 @@ class TestKeyLifecycle: ) @pytest.mark.covers("mgmt.key.update.clear_persists") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.KEY_MANAGEMENT)) def test_explicit_null_clears_the_budget_and_its_reset_time( self, client: ManagementClient, resources: ResourceManager, mock_deployment: str ) -> None: @@ -258,6 +263,14 @@ class TestKeyLifecycle: info, created.written.model_copy(update={"max_budget": None, "budget_duration": None}), replica ) + @meta( + Subject( + domain=Domain.PROXY_AUTH, + providers=(Provider.OPENAI,), + models=(BACKING_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_key_serves_its_model_and_is_denied_others( self, client: ManagementClient, resources: ResourceManager, mock_deployment: str ) -> None: @@ -274,6 +287,15 @@ class TestKeyLifecycle: f"403 body must be a model-access denial, got: {denied.body[:300]}" ) + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.OPENAI,), + models=(BACKING_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_delete_revokes_info_and_chat_on_every_replica( self, client: ManagementClient, resources: ResourceManager, mock_deployment: str ) -> None: diff --git a/tests/e2e/management/test_key_management_e2e.py b/tests/e2e/management/test_key_management_e2e.py index 39a9e657b8c..34f912fe604 100644 --- a/tests/e2e/management/test_key_management_e2e.py +++ b/tests/e2e/management/test_key_management_e2e.py @@ -19,6 +19,7 @@ import pytest from e2e_config import unique_marker from e2e_http import NoBody, StreamingResponse, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from management_client import ManagementClient from models import ( @@ -31,6 +32,7 @@ pytestmark = pytest.mark.e2e TINY_BUDGET = 3e-6 SPEND_MODEL = "claude-haiku-4-5" +SYNTHETIC_BACKEND: Final = "openai/synthetic-detachment" class KeyToggleBlockBody(BaseModel): @@ -161,13 +163,22 @@ def project_resources(client: ManagementClient) -> Iterator[ResourceManager]: class TestKeyManagementRoutes: @pytest.mark.covers("mgmt.key.update.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.OPENAI,), + models=(SYNTHETIC_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_project_detachment_preserves_key_scope_and_refreshes_auth( self, client: ManagementClient, project_resources: ResourceManager ) -> None: resources: Final = project_resources name: Final = f"e2e-detach-{unique_marker()}" model_id: Final = client.proxy.create_model( - name, LiteLLMParamsBody(model="openai/synthetic-detachment", api_key="synthetic", mock_response="orbit") + name, LiteLLMParamsBody(model=SYNTHETIC_BACKEND, api_key="synthetic", mock_response="orbit") ) resources.defer(lambda: client.proxy.delete_model(model_id)) org_id: Final = client.create_org(OrgNewBody(organization_alias=name, models=[name])) @@ -222,6 +233,7 @@ class TestKeyManagementRoutes: assert denied.status_code in (401, 403), denied.body @pytest.mark.covers("mgmt.key.info.persists") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.KEY_MANAGEMENT)) def test_info_reflects_the_fields_the_key_was_created_with( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -246,6 +258,7 @@ class TestKeyManagementRoutes: assert info.rpm_limit == 141414, f"/key/info reports rpm_limit {info.rpm_limit}, configured 141414" @pytest.mark.covers("mgmt.key.unblock.persists") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.KEY_MANAGEMENT)) def test_unblock_flips_key_info_blocked_back( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -266,6 +279,7 @@ class TestKeyManagementRoutes: ) @pytest.mark.covers("mgmt.key.health.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.KEY_MANAGEMENT)) def test_health_reports_the_calling_key_healthy( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -285,6 +299,7 @@ class TestKeyManagementRoutes: ) @pytest.mark.covers("mgmt.key.bulk_update.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.KEY_MANAGEMENT)) def test_bulk_update_applies_max_budget_to_target_key( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -314,6 +329,15 @@ class TestKeyManagementRoutes: ) @pytest.mark.covers("other.key_mgmt.spend_reset.resets_to_value") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.KEY_MANAGEMENT, + providers=(Provider.ANTHROPIC,), + models=(SPEND_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_reset_spend_zeroes_recorded_spend_and_lifts_the_budget_block( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -340,6 +364,7 @@ class TestKeyManagementRoutes: _ = _poll(client, call_allowed_again, "the key stayed budget-blocked after its spend was reset to 0") @pytest.mark.covers("mgmt.key.generate.admin_only") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.KEY_MANAGEMENT)) def test_generate_forbidden_for_non_admin_key( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -355,6 +380,7 @@ class TestKeyManagementRoutes: ) @pytest.mark.covers("mgmt.key.delete.admin_only") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.KEY_MANAGEMENT)) def test_delete_forbidden_for_non_admin_key( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -374,6 +400,7 @@ class TestKeyManagementRoutes: ) @pytest.mark.covers("mgmt.key.update.admin_only") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.KEY_MANAGEMENT)) def test_update_forbidden_for_non_admin_key( self, client: ManagementClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/management/test_management_e2e.py b/tests/e2e/management/test_management_e2e.py index 908eb752611..c167a1323cb 100644 --- a/tests/e2e/management/test_management_e2e.py +++ b/tests/e2e/management/test_management_e2e.py @@ -18,6 +18,7 @@ import pytest from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, UI_PASSWORD, UI_USERNAME, unique_marker from e2e_http import StreamingResponse, Success, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from management_client import ( DASHBOARD_SESSION_TEAM_ID, @@ -47,6 +48,8 @@ from proxy_client import Converged, await_converged pytestmark = pytest.mark.e2e +GEMINI_MODEL: Final = "gemini-2.5-flash" +OPENAI_MODEL: Final = "gpt-5.5" REGENERATE_GRACE_PERIOD = "15s" REGENERATE_GRACE_SECONDS = 15.0 TEAM_DELETE_POOL_OVERFLOW_MEMBERS = 250 @@ -130,6 +133,15 @@ def _poll_model_access_granted(client: ManagementClient, key: str, model: str) - class TestKeyRoutes: @pytest.mark.covers("mgmt.key.generate.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_generate_persists_to_key_info_and_scopes_chat( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -137,12 +149,12 @@ class TestKeyRoutes: key = _generate_key( client, resources, - KeyGenerateBody(models=["gemini-2.5-flash"], key_alias=alias, tpm_limit=424242, rpm_limit=424243), + KeyGenerateBody(models=[GEMINI_MODEL], key_alias=alias, tpm_limit=424242, rpm_limit=424243), ) info = client.proxy.key_info(key) assert info.key_alias == alias, f"/key/info reports key_alias {info.key_alias!r}, configured {alias!r}" - assert info.models == ["gemini-2.5-flash"], ( + assert info.models == [GEMINI_MODEL], ( f"/key/info reports models {info.models}, configured ['gemini-2.5-flash']" ) assert info.tpm_limit == 424242, ( @@ -152,49 +164,73 @@ class TestKeyRoutes: f"/key/info reports rpm_limit {info.rpm_limit}, configured 424243" ) - _poll_chat_ok(client, key, "gemini-2.5-flash") + _poll_chat_ok(client, key, GEMINI_MODEL) _assert_model_denied( - client.chat_status(key, "gpt-5.5", f"say hi {unique_marker()}"), "gpt-5.5" + client.chat_status(key, OPENAI_MODEL, f"say hi {unique_marker()}"), OPENAI_MODEL ) @pytest.mark.covers("mgmt.key.update.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.GEMINI, Provider.OPENAI), + models=(GEMINI_MODEL, OPENAI_MODEL), + mode=Mode.NONSTREAM, + ) + ) def test_update_models_persists_and_flips_enforcement( self, client: ManagementClient, resources: ResourceManager ) -> None: - key = _generate_key(client, resources, KeyGenerateBody(models=["gemini-2.5-flash"])) - _poll_chat_ok(client, key, "gemini-2.5-flash") + key = _generate_key(client, resources, KeyGenerateBody(models=[GEMINI_MODEL])) + _poll_chat_ok(client, key, GEMINI_MODEL) _assert_model_denied( - client.chat_status(key, "gpt-5.5", f"say hi {unique_marker()}"), "gpt-5.5" + client.chat_status(key, OPENAI_MODEL, f"say hi {unique_marker()}"), OPENAI_MODEL ) - client.update_key_models(key, ["gpt-5.5"]) + client.update_key_models(key, [OPENAI_MODEL]) info = client.proxy.key_info(key) - assert info.models == ["gpt-5.5"], ( + assert info.models == [OPENAI_MODEL], ( f"/key/info reports models {info.models} after /key/update to ['gpt-5.5']" ) - _poll_model_access_granted(client, key, "gpt-5.5") - _poll_chat_denied(client, key, "gemini-2.5-flash") + _poll_model_access_granted(client, key, OPENAI_MODEL) + _poll_chat_denied(client, key, GEMINI_MODEL) @pytest.mark.covers("mgmt.key.delete.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_delete_revokes_the_key_on_chat(self, client: ManagementClient, resources: ResourceManager) -> None: """The teardown's deferred delete fires again on the already-deleted key by design: the deferred cleanup must survive this test failing before the in-body delete, and a repeat /key/delete is a cheap no-op the warn-only teardown absorbs.""" - key = _generate_key(client, resources, KeyGenerateBody(models=["gemini-2.5-flash"])) - _poll_chat_ok(client, key, "gemini-2.5-flash") + key = _generate_key(client, resources, KeyGenerateBody(models=[GEMINI_MODEL])) + _poll_chat_ok(client, key, GEMINI_MODEL) client.delete_key_strict(key) def rejected() -> bool | None: - outcome = client.chat_status(key, "gemini-2.5-flash", f"say hi {unique_marker()}") + outcome = client.chat_status(key, GEMINI_MODEL, f"say hi {unique_marker()}") return True if outcome.status_code == 401 else None _ = _poll(client, rejected, "deleted key was still accepted on chat (never rejected 401) at the deadline") @pytest.mark.covers("mgmt.key.list.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + ) + ) def test_created_key_appears_in_key_list_inventory( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -214,8 +250,14 @@ class TestKeyRoutes: @pytest.mark.covers("mgmt.key.block.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + ) + ) def test_block_persists_to_key_info(self, client: ManagementClient, resources: ResourceManager) -> None: - key = _generate_key(client, resources, KeyGenerateBody(models=["gemini-2.5-flash"])) + key = _generate_key(client, resources, KeyGenerateBody(models=[GEMINI_MODEL])) assert not client.proxy.key_info(key).blocked, "/key/info reports the key blocked before /key/block ran" client.block_key(key) @@ -233,6 +275,15 @@ class TestDashboardKeyRoutes: are the same routes the API-surface tests cover with a different caller.""" @pytest.mark.covers("mgmt.key.generate.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.GEMINI,), + models=(GEMINI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_creating_a_key_from_the_dashboard_persists_and_works( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -260,7 +311,7 @@ class TestDashboardKeyRoutes: def dashboard_creates_the_key() -> str | None: match client.generate_key( - KeyGenerateBody(models=["gemini-2.5-flash"], key_alias=alias, tpm_limit=100), + KeyGenerateBody(models=[GEMINI_MODEL], key_alias=alias, tpm_limit=100), caller_key=session.session_key, ): case Success(data=created): @@ -280,7 +331,7 @@ class TestDashboardKeyRoutes: f"/key/info reports key_alias {created_info.key_alias!r} for the key the dashboard created, " f"expected {alias!r}" ) - assert created_info.models == ["gemini-2.5-flash"], ( + assert created_info.models == [GEMINI_MODEL], ( f"/key/info reports models {created_info.models} for the key the dashboard created" ) assert created_info.tpm_limit == 100, ( @@ -301,10 +352,19 @@ class TestDashboardKeyRoutes: "would render no keys", ) - _poll_chat_ok(client, created, "gemini-2.5-flash") - _assert_model_denied(client.chat_status(created, "gpt-5.5", f"say hi {unique_marker()}"), "gpt-5.5") + _poll_chat_ok(client, created, GEMINI_MODEL) + _assert_model_denied(client.chat_status(created, OPENAI_MODEL, f"say hi {unique_marker()}"), OPENAI_MODEL) @pytest.mark.covers("mgmt.key.update.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.GEMINI, Provider.OPENAI), + models=(GEMINI_MODEL, OPENAI_MODEL), + mode=Mode.NONSTREAM, + ) + ) def test_editing_a_key_from_the_dashboard_persists_and_is_enforced( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -312,17 +372,17 @@ class TestDashboardKeyRoutes: target = _generate_key( client, resources, - KeyGenerateBody(models=["gemini-2.5-flash"], key_alias=alias, tpm_limit=100, rpm_limit=200), + KeyGenerateBody(models=[GEMINI_MODEL], key_alias=alias, tpm_limit=100, rpm_limit=200), ) - _poll_chat_ok(client, target, "gemini-2.5-flash") - _assert_model_denied(client.chat_status(target, "gpt-5.5", f"say hi {unique_marker()}"), "gpt-5.5") + _poll_chat_ok(client, target, GEMINI_MODEL) + _assert_model_denied(client.chat_status(target, OPENAI_MODEL, f"say hi {unique_marker()}"), OPENAI_MODEL) session = client.dashboard_login(UI_USERNAME, UI_PASSWORD) resources.defer(lambda: client.proxy.delete_key(session.session_key)) def dashboard_saves_the_edit() -> bool | None: match client.update_key( - KeyUpdateBody(key=target, models=["gpt-5.5"], tpm_limit=300, rpm_limit=400), + KeyUpdateBody(key=target, models=[OPENAI_MODEL], tpm_limit=300, rpm_limit=400), caller_key=session.session_key, ): case Success(): @@ -337,7 +397,7 @@ class TestDashboardKeyRoutes: ) info = client.proxy.key_info(target) - assert info.models == ["gpt-5.5"], ( + assert info.models == [OPENAI_MODEL], ( f"/key/info reports models {info.models} after the dashboard edit to ['gpt-5.5']" ) assert info.tpm_limit == 300, f"/key/info reports tpm_limit {info.tpm_limit} after the dashboard edit to 300" @@ -346,29 +406,38 @@ class TestDashboardKeyRoutes: f"the dashboard edit renamed the key to {info.key_alias!r}, it should still be {alias!r}" ) - _poll_model_access_granted(client, target, "gpt-5.5") - _poll_chat_denied(client, target, "gemini-2.5-flash") + _poll_model_access_granted(client, target, OPENAI_MODEL) + _poll_chat_denied(client, target, GEMINI_MODEL) class TestKeyRegeneration: @pytest.mark.covers("mgmt.key.regenerate.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.OPENAI,), + models=(OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_regenerate_rotates_to_a_working_new_key( self, client: ManagementClient, resources: ResourceManager ) -> None: - old_key = _generate_key(client, resources, KeyGenerateBody(models=["gpt-5.5"])) + old_key = _generate_key(client, resources, KeyGenerateBody(models=[OPENAI_MODEL])) new_key = client.regenerate_key(old_key) resources.defer(lambda: client.proxy.delete_key(new_key)) assert new_key != old_key, "regenerate returned the same key string, so no rotation happened" def new_accepted() -> bool | None: - outcome = client.chat_status(new_key, "gpt-5.5", f"say hi {unique_marker()}") + outcome = client.chat_status(new_key, OPENAI_MODEL, f"say hi {unique_marker()}") return True if outcome.status_code != 401 else None _ = _poll(client, new_accepted, "regenerated key was never accepted at auth (still 401) at the deadline") def old_rejected() -> bool | None: - outcome = client.chat_status(old_key, "gpt-5.5", f"say hi {unique_marker()}") + outcome = client.chat_status(old_key, OPENAI_MODEL, f"say hi {unique_marker()}") return True if outcome.status_code == 401 else None _ = _poll( @@ -376,10 +445,19 @@ class TestKeyRegeneration: ) @pytest.mark.covers("other.key_mgmt.regenerate.grace_period_honored") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + providers=(Provider.OPENAI,), + models=(OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_regenerate_with_grace_period_keeps_old_key_until_revoked( self, client: ManagementClient, resources: ResourceManager ) -> None: - old_key = _generate_key(client, resources, KeyGenerateBody(models=["gpt-5.5"])) + old_key = _generate_key(client, resources, KeyGenerateBody(models=[OPENAI_MODEL])) new_key = client.regenerate_key(old_key, grace_period=REGENERATE_GRACE_PERIOD) resources.defer(lambda: client.proxy.delete_key(new_key)) @@ -387,7 +465,7 @@ class TestKeyRegeneration: assert new_key != old_key, "regenerate returned the same key string, so no rotation happened" def old_accepted() -> bool | None: - outcome = client.chat_status(old_key, "gpt-5.5", f"say hi {unique_marker()}") + outcome = client.chat_status(old_key, OPENAI_MODEL, f"say hi {unique_marker()}") return True if outcome.ok else None _ = _poll(client, old_accepted, "old key was rejected 401 inside its grace period at the deadline") @@ -396,7 +474,7 @@ class TestKeyRegeneration: ) def old_rejected() -> bool | None: - outcome = client.chat_status(old_key, "gpt-5.5", f"say hi {unique_marker()}") + outcome = client.chat_status(old_key, OPENAI_MODEL, f"say hi {unique_marker()}") return True if outcome.status_code == 401 else None _ = _poll( @@ -408,15 +486,21 @@ class TestKeyRegeneration: class TestTeamRoutes: @pytest.mark.covers("mgmt.team.new.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TEAM_MANAGEMENT, + ) + ) def test_new_persists_to_team_info_and_binds_keys( self, client: ManagementClient, resources: ResourceManager ) -> None: alias = f"e2e-mgmt-team-{unique_marker()}" - team_id = _create_team(client, resources, alias, ["gemini-2.5-flash"]) + team_id = _create_team(client, resources, alias, [GEMINI_MODEL]) info = client.team_info(team_id) assert info.team_alias == alias, f"/team/info reports team_alias {info.team_alias!r}, configured {alias!r}" - assert info.models == ["gemini-2.5-flash"], ( + assert info.models == [GEMINI_MODEL], ( f"/team/info reports models {info.models}, configured ['gemini-2.5-flash']" ) @@ -427,8 +511,14 @@ class TestTeamRoutes: ) @pytest.mark.covers("mgmt.team.update.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TEAM_MANAGEMENT, + ) + ) def test_update_persists_to_team_info(self, client: ManagementClient, resources: ResourceManager) -> None: - team_id = _create_team(client, resources, f"e2e-mgmt-team-{unique_marker()}", ["gemini-2.5-flash"]) + team_id = _create_team(client, resources, f"e2e-mgmt-team-{unique_marker()}", [GEMINI_MODEL]) updated_alias = f"e2e-mgmt-team-updated-{unique_marker()}" client.update_team(TeamUpdateBody(team_id=team_id, team_alias=updated_alias)) @@ -438,11 +528,17 @@ class TestTeamRoutes: _ = _poll(client, reflected, f"/team/info never reflected team_alias {updated_alias!r} after /team/update") @pytest.mark.covers("mgmt.team.list.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TEAM_MANAGEMENT, + ) + ) def test_created_team_appears_in_team_list( self, client: ManagementClient, resources: ResourceManager ) -> None: alias = f"e2e-mgmt-team-{unique_marker()}" - team_id = _create_team(client, resources, alias, ["gemini-2.5-flash"]) + team_id = _create_team(client, resources, alias, [GEMINI_MODEL]) _ = _poll( client, @@ -451,17 +547,26 @@ class TestTeamRoutes: ) @pytest.mark.covers("mgmt.team.delete.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TEAM_MANAGEMENT, + providers=(Provider.OPENAI,), + models=(OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_delete_persists_and_revokes_team_bound_key( self, client: ManagementClient, resources: ResourceManager ) -> None: """The teardown's deferred delete_team/delete_key fire again on the already- deleted team and key by design: both are warn-only no-ops, and the deferred cleanup must survive this test failing before the in-body delete.""" - team_id = _create_team(client, resources, f"e2e-mgmt-team-{unique_marker()}", ["gpt-5.5"]) + team_id = _create_team(client, resources, f"e2e-mgmt-team-{unique_marker()}", [OPENAI_MODEL]) key = _generate_key(client, resources, KeyGenerateBody(team_id=team_id)) def accepted() -> bool | None: - outcome = client.chat_status(key, "gpt-5.5", f"say hi {unique_marker()}") + outcome = client.chat_status(key, OPENAI_MODEL, f"say hi {unique_marker()}") return True if outcome.status_code != 401 else None _ = _poll(client, accepted, "team-bound key was never accepted at auth before team deletion") @@ -474,7 +579,7 @@ class TestTeamRoutes: ) def rejected() -> bool | None: - outcome = client.chat_status(key, "gpt-5.5", f"say hi {unique_marker()}") + outcome = client.chat_status(key, OPENAI_MODEL, f"say hi {unique_marker()}") return True if outcome.status_code == 401 else None _ = _poll( @@ -482,6 +587,12 @@ class TestTeamRoutes: ) @pytest.mark.covers("mgmt.team.delete.membership_larger_than_db_pool") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TEAM_MANAGEMENT, + ) + ) def test_team_delete_succeeds_for_team_larger_than_db_pool( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -519,6 +630,12 @@ class TestTeamRoutes: ) @pytest.mark.covers("mgmt.team.member_add.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TEAM_MANAGEMENT, + ) + ) def test_member_add_and_delete_persist_to_team_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -527,7 +644,7 @@ class TestTeamRoutes: resources, UserNewBody(user_email=f"e2e-mgmt-{unique_marker()}@example.com", user_role="internal_user"), ) - team_id = _create_team(client, resources, f"e2e-mgmt-team-{unique_marker()}", ["gemini-2.5-flash"]) + team_id = _create_team(client, resources, f"e2e-mgmt-team-{unique_marker()}", [GEMINI_MODEL]) client.add_team_member(team_id, user_id) member = next( @@ -545,6 +662,12 @@ class TestTeamRoutes: class TestUserRoutes: @pytest.mark.covers("mgmt.user.new.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.USER_MANAGEMENT, + ) + ) def test_new_persists_to_user_info(self, client: ManagementClient, resources: ResourceManager) -> None: email = f"e2e-mgmt-{unique_marker()}@example.com" user_id = _create_user(client, resources, UserNewBody(user_email=email, user_role="internal_user")) @@ -556,6 +679,12 @@ class TestUserRoutes: ) @pytest.mark.covers("mgmt.user.update.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.USER_MANAGEMENT, + ) + ) def test_update_persists_to_user_info(self, client: ManagementClient, resources: ResourceManager) -> None: email = f"e2e-mgmt-{unique_marker()}@example.com" user_id = _create_user(client, resources, UserNewBody(user_email=email, user_role="internal_user")) @@ -572,6 +701,12 @@ class TestUserRoutes: f"/user/info reports user_role {info.user_role!r} after /user/update to 'internal_user_viewer'" ) @pytest.mark.covers("mgmt.user.delete.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.USER_MANAGEMENT, + ) + ) def test_delete_removes_the_user_from_inventory( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -594,6 +729,12 @@ class TestUserRoutes: _ = _poll(client, removed, f"user {user_id} still present in /user/list after /user/delete at the deadline") @pytest.mark.covers("mgmt.user.list.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.USER_MANAGEMENT, + ) + ) def test_created_users_appear_in_user_list( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -616,22 +757,34 @@ class TestUserRoutes: class TestOrganizationRoutes: @pytest.mark.covers("mgmt.organization.new.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.ORGANIZATION_MANAGEMENT, + ) + ) def test_new_persists_to_organization_info( self, client: ManagementClient, resources: ResourceManager ) -> None: alias = f"e2e-mgmt-org-{unique_marker()}" - org_id = client.create_org(OrgNewBody(organization_alias=alias, models=["gemini-2.5-flash"])) + org_id = client.create_org(OrgNewBody(organization_alias=alias, models=[GEMINI_MODEL])) resources.defer(lambda: client.delete_org(org_id)) info = client.org_info(org_id) assert info.organization_alias == alias, ( f"/organization/info reports alias {info.organization_alias!r}, configured {alias!r}" ) - assert info.models == ["gemini-2.5-flash"], ( + assert info.models == [GEMINI_MODEL], ( f"/organization/info reports models {info.models}, configured ['gemini-2.5-flash']" ) @pytest.mark.covers("mgmt.organization.update.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.ORGANIZATION_MANAGEMENT, + ) + ) def test_update_alias_persists_to_organization_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -650,6 +803,12 @@ class TestOrganizationRoutes: ) @pytest.mark.covers("mgmt.organization.delete.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.ORGANIZATION_MANAGEMENT, + ) + ) def test_delete_removes_from_organization_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -674,6 +833,12 @@ class TestOrganizationRoutes: class TestTagRoutes: @pytest.mark.covers("mgmt.tag.new.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TAG_MANAGEMENT, + ) + ) def test_new_persists_to_tag_list(self, client: ManagementClient, resources: ResourceManager) -> None: name = f"e2e-mgmt-tag-{unique_marker()}" description = "Tag for spend categorization" @@ -704,6 +869,12 @@ def _model_entry(client: ManagementClient, model_name: str) -> ModelInfoEntry | class TestModelRoutes: @pytest.mark.covers("mgmt.model.update.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_update_persists_input_cost_to_model_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -747,6 +918,12 @@ class TestModelRoutes: ) @pytest.mark.covers("mgmt.model.delete.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_delete_removes_from_model_info_catalog( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -769,6 +946,12 @@ class TestModelRoutes: _ = _poll(client, absent, f"{model_name} still present in /model/info after /model/delete at the deadline") @pytest.mark.covers("mgmt.model.add.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_new_persists_to_model_info_catalog( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -797,6 +980,11 @@ def _assert_route_forbidden(route: str, outcome: StreamingResponse) -> None: class TestManagementRoutePermissions: @pytest.mark.covers("other.auth.virtual_key.route_permission_enforced") + @meta( + Subject( + domain=Domain.PROXY_AUTH, + ) + ) def test_llm_only_key_forbidden_from_management_writes( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -832,6 +1020,12 @@ class TestManagementRoutePermissions: class TestCustomer: @pytest.mark.covers("mgmt.end_user.new.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.CUSTOMER_MANAGEMENT, + ) + ) def test_customer_create_persists_to_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -896,6 +1090,12 @@ def _generate_response( class TestKeyDeletionAuditLog: @pytest.mark.covers("mgmt.key.delete.audit_logged") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + ) + ) def test_key_delete_by_key_writes_audit_row( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -908,6 +1108,12 @@ class TestKeyDeletionAuditLog: _assert_single_deleted_row(_await_deleted_audit_rows(client, token), token) @pytest.mark.covers("mgmt.key.delete.audit_logged") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.KEY_MANAGEMENT, + ) + ) def test_key_delete_by_alias_writes_audit_row( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -921,6 +1127,12 @@ class TestKeyDeletionAuditLog: _assert_single_deleted_row(_await_deleted_audit_rows(client, token), token) @pytest.mark.covers("mgmt.team.member_delete.audit_logs_keys") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TEAM_MANAGEMENT, + ) + ) def test_team_member_delete_writes_audit_row_for_member_keys( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -940,6 +1152,12 @@ class TestKeyDeletionAuditLog: _assert_single_deleted_row(_await_deleted_audit_rows(client, token), token) @pytest.mark.covers("mgmt.team.delete.audit_logs_keys") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TEAM_MANAGEMENT, + ) + ) def test_team_delete_writes_audit_row_for_team_keys( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -953,6 +1171,12 @@ class TestKeyDeletionAuditLog: _assert_single_deleted_row(_await_deleted_audit_rows(client, token), token) @pytest.mark.covers("mgmt.user.delete.audit_logs_keys") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.USER_MANAGEMENT, + ) + ) def test_user_delete_writes_audit_row_for_user_keys( self, client: ManagementClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/management/test_mcp_lifecycle_e2e.py b/tests/e2e/management/test_mcp_lifecycle_e2e.py index 9257d697647..ca2e99cad8b 100644 --- a/tests/e2e/management/test_mcp_lifecycle_e2e.py +++ b/tests/e2e/management/test_mcp_lifecycle_e2e.py @@ -19,6 +19,7 @@ from typing import Final import pytest from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager from management_client import ManagementClient from models import ( @@ -88,6 +89,12 @@ def _listed_server_everywhere(client: ManagementClient, server_id: str) -> Mappi class TestMcpServerLifecycle: @pytest.mark.covers("mgmt.mcp_server.new.persists") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_create_persists_every_field_on_every_replica( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -107,6 +114,12 @@ class TestMcpServerLifecycle: ) ) @pytest.mark.covers("mgmt.mcp_server.list.persists") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_created_server_is_listed_with_every_field( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -116,6 +129,12 @@ class TestMcpServerLifecycle: _assert_server_matches(row, body, where=f"GET /v1/mcp/server on {replica}") @pytest.mark.covers("mgmt.mcp_server.update.preserves_unrelated_fields") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_updating_only_the_alias_keeps_every_other_field_on_every_replica( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -133,6 +152,12 @@ class TestMcpServerLifecycle: ) @pytest.mark.covers("mgmt.mcp_server.update.clear_persists") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_clearing_the_description_with_null_reads_back_null( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -149,6 +174,12 @@ class TestMcpServerLifecycle: ) @pytest.mark.covers("mgmt.mcp_server.delete.persists") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_delete_removes_the_server_from_every_replica( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -199,6 +230,12 @@ def _toolset_everywhere( class TestMcpToolsetLifecycle: @pytest.mark.covers("mgmt.mcp_toolset.new.persists") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_create_persists_both_tools_under_the_exact_names_written( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -221,6 +258,12 @@ class TestMcpToolsetLifecycle: ) @pytest.mark.covers("mgmt.mcp_toolset.update.preserves_unrelated_fields") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_updating_only_the_description_keeps_the_tools_and_name( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -238,6 +281,12 @@ class TestMcpToolsetLifecycle: ) @pytest.mark.covers("mgmt.mcp_toolset.update.persists") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_updating_the_tools_to_one_entry_reads_back_exactly_that_entry( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -256,6 +305,12 @@ class TestMcpToolsetLifecycle: ) @pytest.mark.covers("mgmt.mcp_toolset.update.clear_persists") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_clearing_the_description_with_null_reads_back_null( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -273,6 +328,12 @@ class TestMcpToolsetLifecycle: ) @pytest.mark.covers("mgmt.mcp_toolset.delete.persists") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_delete_removes_the_toolset_from_every_replica( self, client: ManagementClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/management/test_model_tag_accessgroup_e2e.py b/tests/e2e/management/test_model_tag_accessgroup_e2e.py index eb3a6093c69..5c104db18c1 100644 --- a/tests/e2e/management/test_model_tag_accessgroup_e2e.py +++ b/tests/e2e/management/test_model_tag_accessgroup_e2e.py @@ -22,6 +22,7 @@ from pydantic import BaseModel, ConfigDict, RootModel from e2e_config import unique_marker from e2e_http import NoBody, unwrap +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager from management_client import ManagementClient from models import KeyGenerateBody, LiteLLMParamsBody, ModelInfoBody, ModelNewBody @@ -216,6 +217,12 @@ def _model_blocked_flag(client: ManagementClient, model_id: str) -> bool | None: class TestModelRoutes: @pytest.mark.covers("mgmt.model.add.admin_only") + @meta( + Subject( + domain=Domain.PROXY_AUTH, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_non_admin_key_cannot_add_global_model( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -248,6 +255,12 @@ class TestModelRoutes: ) @pytest.mark.covers("mgmt.model.block.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_block_then_unblock_persists_to_model_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -281,6 +294,12 @@ class TestModelRoutes: class TestTagRoutes: @pytest.mark.covers("mgmt.tag.list.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TAG_MANAGEMENT, + ) + ) def test_tag_list_reports_created_tag(self, client: ManagementClient, resources: ResourceManager) -> None: name = f"e2e-mgmt-tag-{unique_marker()}" description = "coverage: tag inventory" @@ -301,6 +320,12 @@ class TestTagRoutes: ) @pytest.mark.covers("mgmt.tag.delete.persists") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.TAG_MANAGEMENT, + ) + ) def test_tag_delete_removes_from_list(self, client: ManagementClient, resources: ResourceManager) -> None: """The teardown's deferred delete fires again on the already-deleted tag by design: it is the safety net if this test fails before the in-body delete, @@ -326,6 +351,12 @@ class TestTagRoutes: class TestModelAccessGroupRoutes: @pytest.mark.covers("mgmt.access_group.new.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_new_access_group_tags_the_deployment( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -356,6 +387,12 @@ class TestModelAccessGroupRoutes: ) @pytest.mark.covers("mgmt.access_group.info.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.MODEL_MANAGEMENT, + ) + ) def test_access_group_info_reports_membership( self, client: ManagementClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/management/test_model_test_connection_e2e.py b/tests/e2e/management/test_model_test_connection_e2e.py index 25b0b4f24e6..b0d788d582c 100644 --- a/tests/e2e/management/test_model_test_connection_e2e.py +++ b/tests/e2e/management/test_model_test_connection_e2e.py @@ -23,6 +23,7 @@ import time import pytest from e2e_http import unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from management_client import ManagementClient from models import ConnectionTestBody, ConnectionTestResponse, LiteLLMParamsBody @@ -50,6 +51,15 @@ def _probe_mantle(client: ManagementClient) -> ConnectionTestResponse: class TestModelTestConnection: @pytest.mark.covers("mgmt.model.test_connection.happy_path") + @meta( + Subject( + domain=Domain.MANAGEMENT, + route=Route.HEALTH, + providers=(Provider.BEDROCK_MANTLE,), + models=(MANTLE_RESPONSES_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_mantle_responses_connection_succeeds(self, client: ManagementClient) -> None: for attempt in range(1, PROBE_ATTEMPTS + 1): response = _probe_mantle(client) diff --git a/tests/e2e/management/test_team_management_e2e.py b/tests/e2e/management/test_team_management_e2e.py index f30dc6990a9..68a26838ebd 100644 --- a/tests/e2e/management/test_team_management_e2e.py +++ b/tests/e2e/management/test_team_management_e2e.py @@ -27,6 +27,7 @@ import pytest from pydantic import BaseModel from e2e_config import settle_propagation, unique_marker +from e2e_metadata import Domain, Route, Subject, meta from e2e_http import NoBody, PartialBody, StreamingResponse, unwrap from lifecycle import ResourceManager from management_client import ManagementClient @@ -241,6 +242,7 @@ def _member_delete_status(client: ManagementClient, key: str, team_id: str, user class TestTeamManagementRoutes: @pytest.mark.covers("mgmt.team.info.happy_path") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.TEAM_MANAGEMENT)) def test_info_returns_created_team_fields( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -257,6 +259,7 @@ class TestTeamManagementRoutes: ) @pytest.mark.covers("mgmt.team.block.persists") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.TEAM_MANAGEMENT)) def test_block_then_unblock_persists_to_team_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -278,6 +281,7 @@ class TestTeamManagementRoutes: ) @pytest.mark.covers("mgmt.team.member_update.persists") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.TEAM_MANAGEMENT)) def test_member_update_persists_role_and_budget( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -302,6 +306,7 @@ class TestTeamManagementRoutes: ) @pytest.mark.covers("mgmt.team.member_delete.persists") + @meta(Subject(domain=Domain.MANAGEMENT, route=Route.TEAM_MANAGEMENT)) def test_member_delete_persists_to_team_info( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -320,6 +325,7 @@ class TestTeamManagementRoutes: ) @pytest.mark.covers("mgmt.team.new.admin_only") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_new_is_denied_to_non_admin_keys( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -334,6 +340,7 @@ class TestTeamManagementRoutes: ) @pytest.mark.covers("mgmt.team.member_add.member_forbidden") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_member_add_forbidden_to_plain_member( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -348,6 +355,7 @@ class TestTeamManagementRoutes: ) @pytest.mark.covers("mgmt.team.member_delete.member_forbidden") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_member_delete_forbidden_to_plain_member( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -458,6 +466,7 @@ class TestTeamAdminWithNoEditableFields: """No proxy admin has enabled a team field for team admins, which is how every proxy starts.""" @pytest.mark.covers("mgmt.team.update.team_admin_forbidden_until_enabled") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_team_admin_cannot_change_any_team_setting( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -483,6 +492,7 @@ class TestTeamAdminWithTpmLimitEnabled: """A proxy admin has enabled tpm_limit, so a team admin may change that setting and no other.""" @pytest.mark.covers("mgmt.team.update.team_admin_limited_to_enabled_fields") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_team_admin_saves_the_settings_form_with_a_new_tpm_limit( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -522,6 +532,7 @@ class TestTeamAdminWithTpmLimitEnabled: pytest.param(TeamSettingsChange(metadata=TeamCustomMetadata(cost_center="team-admin")), id="metadata"), ], ) + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_team_admin_cannot_change_a_setting_that_is_not_enabled( self, client: ManagementClient, resources: ResourceManager, change: TeamSettingsChange ) -> None: @@ -548,6 +559,7 @@ class TestTeamAdminWithTpmLimitEnabled: ) @pytest.mark.covers("mgmt.team.update.team_admin_resend_keeps_budget_reset") + @meta(Subject(domain=Domain.SPEND_BUDGETS, route=Route.TEAM_MANAGEMENT)) def test_team_admin_resending_the_budget_settings_keeps_the_next_budget_reset( self, client: ManagementClient, resources: ResourceManager ) -> None: @@ -613,6 +625,7 @@ class TestTeamAdminWithRpmLimitAndMaxBudgetEnabled: "current_budget", [pytest.param(_TEAM_MAX_BUDGET, id="lower"), pytest.param(None, id="first-budget")], ) + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_team_admin_saves_a_new_rpm_limit_and_a_tighter_budget( self, client: ManagementClient, resources: ResourceManager, current_budget: float | None ) -> None: @@ -649,6 +662,7 @@ class TestTeamAdminWithRpmLimitAndMaxBudgetEnabled: pytest.param(None, "Only a proxy admin can remove", id="remove"), ], ) + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_team_admin_cannot_raise_or_remove_the_budget( self, client: ManagementClient, resources: ResourceManager, max_budget: float | None, refusal: str ) -> None: @@ -670,6 +684,7 @@ class TestTeamAdminWithRpmLimitAndMaxBudgetEnabled: ) @pytest.mark.covers("mgmt.team.update.team_admin_cannot_grow_budget") + @meta(Subject(domain=Domain.PROXY_AUTH, route=Route.TEAM_MANAGEMENT)) def test_team_admin_cannot_raise_an_org_team_budget_under_the_org_cap( self, client: ManagementClient, resources: ResourceManager ) -> None: From 79209b92a1b87c1f205d87a5d65e23a97a4441b5 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Wed, 7 Oct 2026 10:32:25 -0700 Subject: [PATCH 05/13] test(e2e): tag guardrails and logging tests with Subject metadata and record client steps (#44963) * test(e2e): add enum values, auto-discovering label gates and secret hiding for e2e metadata * test(e2e): tag guardrails and logging tests with Subject metadata and record client steps * test(e2e): leave the guardrails and logging harness unit tests untagged * test(e2e): let the inner create_model step name the guardrail backend deployment * docs(e2e): name every markerless harness test file that carries no Subject * test(e2e): keep the step discovery comprehensions to one for clause * test(e2e): declare the default guardrail backend model on the tests that drive it --- tests/e2e/guardrails/guardrails_client.py | 36 +++++- .../guardrails/test_apply_guardrail_e2e.py | 7 ++ .../guardrails/test_bedrock_guardrail_e2e.py | 44 +++++++ ...test_block_code_execution_guardrail_e2e.py | 11 +- .../guardrails/test_guardrail_dispatch_e2e.py | 9 ++ ...test_guardrail_information_response_e2e.py | 23 +++- .../test_key_guardrail_image_edit_e2e.py | 10 ++ .../test_key_guardrail_video_e2e.py | 9 ++ ...t_openai_moderation_category_matrix_e2e.py | 40 ++++++- .../test_openai_moderation_guardrail_e2e.py | 20 ++++ .../test_policy_inherited_guardrail_e2e.py | 17 +++ .../guardrails/test_presidio_masking_e2e.py | 107 +++++++++++++++++ ...est_responses_pre_call_block_stream_e2e.py | 25 +++- .../test_streaming_guardrail_e2e.py | 9 ++ .../test_team_disable_global_guardrail_e2e.py | 17 +++ .../test_tool_permission_guardrail_e2e.py | 19 +++ tests/e2e/logging/datadog_reader.py | 5 + tests/e2e/logging/gcs_reader.py | 5 +- tests/e2e/logging/logging_client.py | 26 ++++ tests/e2e/logging/s3_reader.py | 5 + tests/e2e/logging/test_datadog_log_e2e.py | 79 +++++++++++- tests/e2e/logging/test_gcs_log_e2e.py | 9 ++ .../test_langsmith_batch_serialization_e2e.py | 9 ++ tests/e2e/logging/test_otel_trace_e2e.py | 113 +++++++++++++++++- ..._otel_v2_langfuse_generation_output_e2e.py | 110 +++++++++++++++-- .../test_prometheus_cardinality_e2e.py | 10 ++ .../logging/test_prometheus_queue_time_e2e.py | 10 ++ tests/e2e/logging/test_s3_log_e2e.py | 29 ++++- .../test_team_langfuse_callback_e2e.py | 9 ++ tests/e2e/logging/test_weave_log_e2e.py | 22 +++- tests/e2e/logging/weave_reader.py | 7 +- 31 files changed, 821 insertions(+), 30 deletions(-) diff --git a/tests/e2e/guardrails/guardrails_client.py b/tests/e2e/guardrails/guardrails_client.py index 7772a6a1e85..b193896bf6c 100644 --- a/tests/e2e/guardrails/guardrails_client.py +++ b/tests/e2e/guardrails/guardrails_client.py @@ -11,6 +11,7 @@ from typing import Final, Literal from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, SLOW_PROVIDER_TIMEOUT_SECONDS, settle_propagation, unique_marker from e2e_http import NoBody, Result, StreamingResponse, Success, unwrap +from e2e_metadata import step from lifecycle import ResourceManager from models import ( AnthropicMessagesBody, @@ -35,7 +36,9 @@ from models import ( VideoCreateResponse, ) from proxy_client import ProxyClient -from pydantic import BaseModel +from pydantic import BaseModel, Field + +GUARDRAIL_BACKEND: Final = "gemini/gemini-2.5-flash" GuardrailMode = Literal["pre_call", "post_call", "during_call", "logging_only"] PiiEntity = Literal["EMAIL_ADDRESS", "PHONE_NUMBER", "PERSON", "CREDIT_CARD", "US_SSN"] @@ -62,14 +65,14 @@ class BedrockGuardrailParamsBody(GuardrailParamsBase): guardrail: Literal["bedrock"] = "bedrock" guardrailIdentifier: str guardrailVersion: str - aws_access_key_id: str | None = None - aws_secret_access_key: str | None = None + aws_access_key_id: str | None = Field(default=None, repr=False) + aws_secret_access_key: str | None = Field(default=None, repr=False) aws_region_name: str | None = None class OpenAIModerationParamsBody(GuardrailParamsBase): guardrail: Literal["openai_moderation"] = "openai_moderation" - api_key: str | None = None + api_key: str | None = Field(default=None, repr=False) model: str | None = None @@ -193,6 +196,7 @@ class _ResponsesGuardrailBody(BaseModel): class GuardrailsClient: proxy: ProxyClient + @step("Register the content filter guardrail {name} that blocks prompts containing {blocked_keyword}") def create_content_filter_guardrail(self, name: str, blocked_keyword: str, *, default_on: bool = True) -> str: return self.register( name, @@ -203,6 +207,7 @@ class GuardrailsClient: ), ) + @step("Register the Bedrock guardrail {name}") def create_bedrock_guardrail( self, name: str, @@ -235,7 +240,7 @@ class GuardrailsClient: resources: ResourceManager, prefix: str = "e2e-guard-backend", *, - backend: str = "gemini/gemini-2.5-flash", + backend: str = GUARDRAIL_BACKEND, api_key: str = "os.environ/GEMINI_API_KEY", ) -> str: """Register a chat deployment for a guardrail test to run against @@ -250,6 +255,7 @@ class GuardrailsClient: resources.defer(lambda: self.proxy.delete_model(model_id)) return model_name + @step("Register the {params.guardrail} guardrail {name} with mode {params.mode}") def register(self, name: str, params: GuardrailParamsBody) -> str: """Register any guardrail via POST /guardrails and return its id, once every replica can be expected to serve it. New built-ins register with @@ -274,6 +280,7 @@ class GuardrailsClient: settle_propagation(time.monotonic()) return guardrail_id + @step("Delete the guardrail") def delete_guardrail(self, guardrail_id: str) -> None: _ = self.proxy.transport.delete( f"/guardrails/{guardrail_id}", @@ -282,6 +289,7 @@ class GuardrailsClient: response_type=NoBody, ) + @step("Create the guardrail policy {body.policy_name} that adds the guardrails {body.guardrails_add}") def create_policy(self, body: PolicyCreateBody) -> str: """Create a policy via POST /policies and return its name once every replica can be expected to serve it (policies reach the data plane on the periodic @@ -297,6 +305,7 @@ class GuardrailsClient: settle_propagation(time.monotonic()) return created.policy_name + @step("Delete every version of the guardrail policy {policy_name}") def delete_policy(self, policy_name: str) -> None: _ = self.proxy.transport.delete( f"/policies/name/{policy_name}/all-versions", @@ -305,6 +314,7 @@ class GuardrailsClient: response_type=NoBody, ) + @step("Attach the guardrail policy {policy_name} to requests tagged {tags}") def attach_policy_to_tags(self, policy_name: str, tags: list[str]) -> str: attachment_id = unwrap( self.proxy.transport.post( @@ -317,6 +327,7 @@ class GuardrailsClient: settle_propagation(time.monotonic()) return attachment_id + @step("Delete the guardrail policy attachment") def delete_policy_attachment(self, attachment_id: str) -> None: _ = self.proxy.transport.delete( f"/policies/attachments/{attachment_id}", @@ -325,6 +336,7 @@ class GuardrailsClient: response_type=NoBody, ) + @step("Create the team {alias} opted out of global guardrails and wait until /team/info returns it") def create_team_opted_out_of_global_guardrails(self, alias: str) -> str: team_id = unwrap( self.proxy.transport.post( @@ -340,6 +352,7 @@ class GuardrailsClient: self._await_team(team_id) return team_id + @step("Delete the team") def delete_team(self, team_id: str) -> None: _ = self.proxy.transport.post( "/team/delete", @@ -348,9 +361,11 @@ class GuardrailsClient: response_type=NoBody, ) + @step("Generate a virtual key in the team") def create_key_in_team(self, team_id: str) -> str: return self.proxy.generate_key(KeyGenerateBody(team_id=team_id, user_id="e2e-guardrails-user")) + @step("Generate a virtual key with the guardrails {guardrails}") def create_key_with_guardrails(self, resources: ResourceManager, guardrails: list[str]) -> str: key = self.proxy.generate_key( KeyGenerateBody(user_id="e2e-guardrails-user", metadata=KeyMetadata(guardrails=guardrails)) @@ -358,6 +373,7 @@ class GuardrailsClient: resources.defer(lambda: self.proxy.delete_key(key)) return key + @step("Send a /v1/videos request to {model}") def create_video(self, key: str, model: str, prompt: str) -> Result[VideoCreateResponse]: return self.proxy.transport.post( "/v1/videos", @@ -366,6 +382,7 @@ class GuardrailsClient: response_type=VideoCreateResponse, ) + @step("Send a /v1/images/edits request to {model}") def edit_image(self, key: str, model: str, prompt: str, image: bytes) -> Result[ImageGenerationResponse]: return self.proxy.transport.upload( "/v1/images/edits", @@ -379,6 +396,7 @@ class GuardrailsClient: timeout=SLOW_PROVIDER_TIMEOUT_SECONDS, ) + @step("Send a /chat/completions request to {model}") def chat( self, key: str, @@ -407,6 +425,7 @@ class GuardrailsClient: ), ) + @step("Send a /chat/completions request to {model}") def chat_raw( self, key: str, @@ -438,6 +457,7 @@ class GuardrailsClient: ), ) + @step("Send a streaming /chat/completions request to {model}") def chat_stream_raw( self, key: str, @@ -462,6 +482,7 @@ class GuardrailsClient: ), ) + @step("Send a /v1/messages request to {model}") def messages( self, key: str, @@ -481,6 +502,7 @@ class GuardrailsClient: ), ) + @step("Send a /v1/messages request to {model}") def messages_raw( self, key: str, @@ -501,6 +523,7 @@ class GuardrailsClient: ), ) + @step("Send a streaming /v1/messages request to {model}") def messages_stream_raw( self, key: str, @@ -521,6 +544,7 @@ class GuardrailsClient: ), ) + @step("Send a /v1/responses request to {model}") def responses( self, key: str, @@ -535,6 +559,7 @@ class GuardrailsClient: json=_ResponsesGuardrailBody(model=model, input=text, guardrails=guardrails), ) + @step("Send a streaming /v1/responses request to {model}") def responses_stream_raw( self, key: str, @@ -553,6 +578,7 @@ class GuardrailsClient: stream=True, ) + @step("Apply the guardrail {name} to a piece of text with /guardrails/apply_guardrail") def apply_guardrail(self, key: str, *, name: str, text: str) -> Result[ApplyGuardrailResponse]: return self.proxy.transport.post( "/guardrails/apply_guardrail", diff --git a/tests/e2e/guardrails/test_apply_guardrail_e2e.py b/tests/e2e/guardrails/test_apply_guardrail_e2e.py index ee691db22da..7a4db80e379 100644 --- a/tests/e2e/guardrails/test_apply_guardrail_e2e.py +++ b/tests/e2e/guardrails/test_apply_guardrail_e2e.py @@ -11,6 +11,7 @@ import pytest from e2e_config import MASTER_KEY, unique_marker from e2e_http import Success, UnauthorizedError, UnknownApiError +from e2e_metadata import Domain, Route, Subject, meta from guardrails_client import GuardrailsClient from lifecycle import ResourceManager @@ -23,6 +24,12 @@ class TestApplyGuardrailEndpoint: "guardrail.litellm_content_filter.apply_endpoint.allows", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.GUARDRAILS, + ) + ) def test_apply_guardrail_blocks_banned_and_allows_clean( self, client: GuardrailsClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/guardrails/test_bedrock_guardrail_e2e.py b/tests/e2e/guardrails/test_bedrock_guardrail_e2e.py index aeffec24c61..fac15645f83 100644 --- a/tests/e2e/guardrails/test_bedrock_guardrail_e2e.py +++ b/tests/e2e/guardrails/test_bedrock_guardrail_e2e.py @@ -22,6 +22,7 @@ from typing import Final import pytest from e2e_config import unique_marker from e2e_http import StreamingResponse, UnknownApiError +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from guardrails_client import ( BedrockGuardrailParamsBody, GuardrailsClient, @@ -59,6 +60,15 @@ class TestBedrockGuardrail: "guardrail.bedrock.pre_call.blocks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_pre_call_blocks_harmful_prompt( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -96,6 +106,14 @@ class TestBedrockGuardrail: "guardrail.bedrock.post_call.blocks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_post_call_blocks_denied_model_output( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -138,6 +156,15 @@ class TestBedrockGuardrail: pytest.fail(f"bedrock post_call guardrail did not block denied model output; got {result}") @pytest.mark.covers("guardrail.bedrock.pre_call.blocks", exercised_on=["messages"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.MESSAGES, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_pre_call_blocks_on_messages( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -149,6 +176,15 @@ class TestBedrockGuardrail: _assert_policy_block(result, "/v1/messages") @pytest.mark.covers("guardrail.bedrock.pre_call.blocks", exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.RESPONSES, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_bedrock_pre_call_blocks_on_responses( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -160,6 +196,14 @@ class TestBedrockGuardrail: _assert_policy_block(result, "/v1/responses") @pytest.mark.covers("guardrail.bedrock.post_call.blocks", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_bedrock_post_call_blocks_denied_streamed_output_and_passes_clean_streams( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/guardrails/test_block_code_execution_guardrail_e2e.py b/tests/e2e/guardrails/test_block_code_execution_guardrail_e2e.py index 7cf4c195424..7cda38d9aa4 100644 --- a/tests/e2e/guardrails/test_block_code_execution_guardrail_e2e.py +++ b/tests/e2e/guardrails/test_block_code_execution_guardrail_e2e.py @@ -20,7 +20,8 @@ import pytest from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, unique_marker from e2e_http import unwrap -from guardrails_client import BlockCodeExecutionParamsBody, GuardrailsClient +from e2e_metadata import Domain, Mode, Provider, Subject, meta +from guardrails_client import GUARDRAIL_BACKEND, BlockCodeExecutionParamsBody, GuardrailsClient from lifecycle import ResourceManager from models import ChatResponse @@ -45,6 +46,14 @@ class TestBlockCodeExecutionGuardrail: "guardrail.block_code_execution.pre_call.blocks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(GUARDRAIL_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_blocks_execution_request_but_allows_explanation( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/guardrails/test_guardrail_dispatch_e2e.py b/tests/e2e/guardrails/test_guardrail_dispatch_e2e.py index 793974ccdb1..c8d295ee459 100644 --- a/tests/e2e/guardrails/test_guardrail_dispatch_e2e.py +++ b/tests/e2e/guardrails/test_guardrail_dispatch_e2e.py @@ -11,6 +11,7 @@ from __future__ import annotations import pytest from e2e_config import unique_marker from e2e_http import UnknownApiError, ValidationError +from e2e_metadata import Domain, Mode, Provider, Subject, meta from guardrails_client import GuardrailsClient pytestmark = pytest.mark.e2e @@ -28,6 +29,14 @@ MODEL = "gemini-2.5-flash" "guardrail.dispatch.pre_call.rejects_unknown_name", exercised_on=["chat_completions"], ) +@meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) +) def test_request_naming_an_unknown_guardrail_fails_closed(client: GuardrailsClient, scoped_key: str) -> None: result = client.chat(scoped_key, MODEL, "say hi", guardrails=[f"e2e-no-such-guardrail-{unique_marker()}"]) diff --git a/tests/e2e/guardrails/test_guardrail_information_response_e2e.py b/tests/e2e/guardrails/test_guardrail_information_response_e2e.py index 9701ea35819..e0475610e8f 100644 --- a/tests/e2e/guardrails/test_guardrail_information_response_e2e.py +++ b/tests/e2e/guardrails/test_guardrail_information_response_e2e.py @@ -9,6 +9,7 @@ import pytest from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Mode, Provider, Subject, meta from guardrails_client import ( BlockedWordBody, ContentFilterParamsBody, @@ -19,6 +20,8 @@ from models import ChatResponse, GuardrailInformationEntry pytestmark = pytest.mark.e2e +BACKEND_MODEL: Final = "openai/gpt-4.1-mini" + GUARDRAIL_PROPAGATION_DEADLINE_SECONDS: Final = 40.0 GUARDRAIL_PROPAGATION_POLL_INTERVAL_SECONDS: Final = 5.0 @@ -60,6 +63,14 @@ class TestGuardrailInformationResponse: "guardrail.litellm_content_filter.pre_call.returns_guardrail_information", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.OPENAI,), + models=(BACKEND_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_flag_returns_guardrail_information_for_the_guardrail_that_ran( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -68,7 +79,7 @@ class TestGuardrailInformationResponse: model = client.create_backend_model( resources, prefix="e2e-guardrail-info-backend", - backend="openai/gpt-4.1-mini", + backend=BACKEND_MODEL, api_key="os.environ/OPENAI_API_KEY", ) deadline = time.monotonic() + GUARDRAIL_PROPAGATION_DEADLINE_SECONDS @@ -91,6 +102,14 @@ class TestGuardrailInformationResponse: ) time.sleep(GUARDRAIL_PROPAGATION_POLL_INTERVAL_SECONDS) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.OPENAI,), + models=(BACKEND_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_without_flag_response_has_no_guardrail_information( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -99,7 +118,7 @@ class TestGuardrailInformationResponse: model = client.create_backend_model( resources, prefix="e2e-guardrail-info-backend", - backend="openai/gpt-4.1-mini", + backend=BACKEND_MODEL, api_key="os.environ/OPENAI_API_KEY", ) diff --git a/tests/e2e/guardrails/test_key_guardrail_image_edit_e2e.py b/tests/e2e/guardrails/test_key_guardrail_image_edit_e2e.py index 388a554cde7..153d5fd0aa7 100644 --- a/tests/e2e/guardrails/test_key_guardrail_image_edit_e2e.py +++ b/tests/e2e/guardrails/test_key_guardrail_image_edit_e2e.py @@ -6,6 +6,7 @@ from typing import Final import pytest from e2e_config import unique_marker from e2e_http import Success, UnknownApiError +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from guardrails_client import GuardrailsClient, poll_until_blocked from lifecycle import ResourceManager from models import LiteLLMParamsBody @@ -41,6 +42,15 @@ class TestKeyAttachedGuardrailOnImageEdits: "guardrail.litellm_content_filter.pre_call.blocks_image_edit", exercised_on=["images_edits"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.IMAGES, + providers=(Provider.GEMINI, Provider.OPENAI,), + models=(CHAT_MODEL, IMAGE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_key_attached_content_filter_blocks_banned_image_edit_prompt( self, client: GuardrailsClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/guardrails/test_key_guardrail_video_e2e.py b/tests/e2e/guardrails/test_key_guardrail_video_e2e.py index 5f318e141a5..233cbfb8881 100644 --- a/tests/e2e/guardrails/test_key_guardrail_video_e2e.py +++ b/tests/e2e/guardrails/test_key_guardrail_video_e2e.py @@ -3,6 +3,7 @@ from __future__ import annotations import pytest from e2e_config import unique_marker from e2e_http import Success, UnknownApiError +from e2e_metadata import Domain, Mode, Provider, Subject, meta from guardrails_client import GuardrailsClient, poll_until_blocked from lifecycle import ResourceManager from models import LiteLLMParamsBody @@ -38,6 +39,14 @@ class TestKeyAttachedGuardrailOnVideos: "guardrail.litellm_content_filter.pre_call.blocks_video", exercised_on=["videos"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI, Provider.VERTEX_AI,), + models=(CHAT_MODEL, VIDEO_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_key_attached_content_filter_blocks_banned_video_prompt( self, client: GuardrailsClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/guardrails/test_openai_moderation_category_matrix_e2e.py b/tests/e2e/guardrails/test_openai_moderation_category_matrix_e2e.py index 1f1af818290..10155cf96a5 100644 --- a/tests/e2e/guardrails/test_openai_moderation_category_matrix_e2e.py +++ b/tests/e2e/guardrails/test_openai_moderation_category_matrix_e2e.py @@ -7,15 +7,22 @@ body that names moderation; a refine-wrapper bypass must also be blocked. from __future__ import annotations +from typing import Final + import pytest from e2e_config import unique_marker from e2e_http import Result, UnknownApiError +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from guardrails_client import GuardrailsClient, OpenAIModerationParamsBody from lifecycle import ResourceManager from models import AnthropicMessagesResponse, ChatResponse pytestmark = pytest.mark.e2e +GEMINI_BACKEND: Final = "gemini/gemini-2.5-flash" +ANTHROPIC_BACKEND: Final = "anthropic/claude-haiku-4-5" +OPENAI_BACKEND: Final = "openai/gpt-4o-mini" + CATEGORY_PROMPTS: tuple[tuple[str, str], ...] = ( ( "violence", @@ -79,6 +86,15 @@ class TestOpenAIModerationCategoryMatrix: "guardrail.openai_moderations.pre_call.blocks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(GEMINI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_blocks_category( self, client: GuardrailsClient, @@ -89,7 +105,7 @@ class TestOpenAIModerationCategoryMatrix: client, resources, prefix="e2e-mod-cat-chat", - backend="gemini/gemini-2.5-flash", + backend=GEMINI_BACKEND, api_key="os.environ/GEMINI_API_KEY", ) for category, prompt in CATEGORY_PROMPTS: @@ -99,6 +115,15 @@ class TestOpenAIModerationCategoryMatrix: "guardrail.openai_moderations.pre_call.blocks", exercised_on=["messages"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(ANTHROPIC_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_messages_blocks_category( self, client: GuardrailsClient, @@ -109,7 +134,7 @@ class TestOpenAIModerationCategoryMatrix: client, resources, prefix="e2e-mod-cat-msg", - backend="anthropic/claude-haiku-4-5", + backend=ANTHROPIC_BACKEND, api_key="os.environ/ANTHROPIC_API_KEY", ) for category, prompt in CATEGORY_PROMPTS: @@ -119,6 +144,15 @@ class TestOpenAIModerationCategoryMatrix: "guardrail.openai_moderations.pre_call.blocks", exercised_on=["responses"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_blocks_category( self, client: GuardrailsClient, @@ -129,7 +163,7 @@ class TestOpenAIModerationCategoryMatrix: client, resources, prefix="e2e-mod-cat-resp", - backend="openai/gpt-4o-mini", + backend=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY", ) for category, prompt in CATEGORY_PROMPTS: diff --git a/tests/e2e/guardrails/test_openai_moderation_guardrail_e2e.py b/tests/e2e/guardrails/test_openai_moderation_guardrail_e2e.py index 43deb279bc8..b9593feb2ed 100644 --- a/tests/e2e/guardrails/test_openai_moderation_guardrail_e2e.py +++ b/tests/e2e/guardrails/test_openai_moderation_guardrail_e2e.py @@ -18,7 +18,9 @@ import pytest from e2e_config import unique_marker from e2e_http import UnknownApiError, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from guardrails_client import ( + GUARDRAIL_BACKEND, GuardrailsClient, OpenAIModerationParamsBody, poll_until_blocked, @@ -37,6 +39,15 @@ class TestOpenAIModerationGuardrail: "guardrail.openai_moderations.pre_call.blocks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(GUARDRAIL_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_moderation_blocks_flagged_input( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -76,6 +87,15 @@ class TestOpenAIModerationGuardrail: "guardrail.openai_moderations.pre_call.blocks", exercised_on=["messages"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.MESSAGES, + providers=(Provider.GEMINI,), + models=(GUARDRAIL_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_moderation_blocks_flagged_input_on_messages( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/guardrails/test_policy_inherited_guardrail_e2e.py b/tests/e2e/guardrails/test_policy_inherited_guardrail_e2e.py index 6298a1de038..8c8e86ebe7a 100644 --- a/tests/e2e/guardrails/test_policy_inherited_guardrail_e2e.py +++ b/tests/e2e/guardrails/test_policy_inherited_guardrail_e2e.py @@ -16,6 +16,7 @@ from __future__ import annotations import pytest from e2e_config import CHEAP_OPENAI_MODEL, unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Subject, meta from guardrails_client import ( GuardrailsClient, PolicyConditionBody, @@ -73,6 +74,14 @@ def _setup_child_policy_attached_to_tag( class TestPolicyInheritedGuardrail: + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.OPENAI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_child_condition_miss_still_applies_inherited_parent_guardrail( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -101,6 +110,14 @@ class TestPolicyInheritedGuardrail: f"the child's own guardrail must not run when its condition fails; got {outcome.headers}" ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.OPENAI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_child_condition_match_applies_child_and_inherited_parent_guardrails( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/guardrails/test_presidio_masking_e2e.py b/tests/e2e/guardrails/test_presidio_masking_e2e.py index 35eb4884ddf..9fc2c7d3ea2 100644 --- a/tests/e2e/guardrails/test_presidio_masking_e2e.py +++ b/tests/e2e/guardrails/test_presidio_masking_e2e.py @@ -40,6 +40,7 @@ from pydantic import BaseModel, JsonValue, TypeAdapter from e2e_config import unique_marker from e2e_http import Result, StreamingResponse, Success +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from guardrails_client import GuardrailMode, GuardrailsClient, PiiAction, PiiEntity, PresidioParamsBody from lifecycle import ResourceManager from models import ( @@ -248,6 +249,15 @@ class TestPresidioPreCallMasking: "guardrail.presidio.pre_call.masks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_pre_call_masks_pii_on_chat_completions( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -267,6 +277,15 @@ class TestPresidioPreCallMasking: "guardrail.presidio.pre_call.masks", exercised_on=["messages"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.MESSAGES, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_pre_call_masks_pii_on_messages( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -312,6 +331,14 @@ class TestPresidioPostCallMasking: "guardrail.presidio.post_call.masks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_post_call_masks_pii_in_model_output( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -393,6 +420,15 @@ class TestPresidioCreditCardOutputMasking: "guardrail.presidio.post_call.masks_generated_output", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_ui_default_scope_masks_a_card_number_the_model_generates_on_chat_completions( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -421,6 +457,15 @@ class TestPresidioCreditCardOutputMasking: "guardrail.presidio.post_call.masks_generated_output", exercised_on=["chat_completions_stream"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_ui_default_scope_masks_a_card_number_the_model_generates_on_streaming_chat_completions( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -453,6 +498,15 @@ class TestPresidioCreditCardOutputMasking: "guardrail.presidio.post_call.masks_generated_output", exercised_on=["anthropic_messages_stream"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.MESSAGES, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_ui_default_scope_masks_a_card_number_the_model_generates_on_streaming_anthropic_messages( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -571,6 +625,15 @@ class TestPresidioSpendLogStoresMaskedOutput: ) @pytest.mark.covers(_CELL, exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_spend_log_stores_masked_output_on_chat_completions( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -585,6 +648,15 @@ class TestPresidioSpendLogStoresMaskedOutput: ) @pytest.mark.covers(_CELL, exercised_on=["chat_completions_stream"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_spend_log_stores_masked_output_on_streaming_chat_completions( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -599,6 +671,15 @@ class TestPresidioSpendLogStoresMaskedOutput: ) @pytest.mark.covers(_CELL, exercised_on=["messages"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.MESSAGES, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_spend_log_stores_masked_output_on_anthropic_messages( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -613,6 +694,15 @@ class TestPresidioSpendLogStoresMaskedOutput: ) @pytest.mark.covers(_CELL, exercised_on=["anthropic_messages_stream"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.MESSAGES, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_spend_log_stores_masked_output_on_streaming_anthropic_messages( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -627,6 +717,15 @@ class TestPresidioSpendLogStoresMaskedOutput: ) @pytest.mark.covers(_CELL, exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.RESPONSES, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_spend_log_stores_masked_output_on_responses( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -673,6 +772,14 @@ class TestPresidioSpendLogRecord: "guardrail.presidio.pre_call.logs_masked_entities", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_masking_run_is_recorded_on_the_spend_log( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/guardrails/test_responses_pre_call_block_stream_e2e.py b/tests/e2e/guardrails/test_responses_pre_call_block_stream_e2e.py index 93512e2a64c..cdf3098e85d 100644 --- a/tests/e2e/guardrails/test_responses_pre_call_block_stream_e2e.py +++ b/tests/e2e/guardrails/test_responses_pre_call_block_stream_e2e.py @@ -7,12 +7,15 @@ from typing import Final import pytest from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from guardrails_client import CustomCodeParamsBody, GuardrailsClient from lifecycle import ResourceManager from pydantic import BaseModel, TypeAdapter pytestmark = pytest.mark.e2e +BACKEND_MODEL: Final = "openai/gpt-4.1-mini" + DENIAL: Final = "This model is not currently available. Please contact support if you think this is a mistake." CUSTOM_CODE: Final = f''' @@ -109,12 +112,21 @@ class TestResponsesPreCallBlock: return name @pytest.mark.covers("guardrail.custom_code.pre_call.blocks", exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(BACKEND_MODEL,), + mode=Mode.STREAM, + ) + ) def test_stream_block_is_sse_with_completed_assistant_message( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: name: Final = self._register_block(client, resources) model: Final = client.create_backend_model( - resources, prefix="e2e-responses-block", backend="openai/gpt-4.1-mini", api_key="os.environ/OPENAI_API_KEY" + resources, prefix="e2e-responses-block", backend=BACKEND_MODEL, api_key="os.environ/OPENAI_API_KEY" ) result: Final = _poll_for_block( @@ -137,12 +149,21 @@ class TestResponsesPreCallBlock: _assert_blocked_response(completed[0].response) @pytest.mark.covers("guardrail.custom_code.pre_call.blocks", exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(BACKEND_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_non_stream_block_is_schema_valid_json( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: name: Final = self._register_block(client, resources) model: Final = client.create_backend_model( - resources, prefix="e2e-responses-block", backend="openai/gpt-4.1-mini", api_key="os.environ/OPENAI_API_KEY" + resources, prefix="e2e-responses-block", backend=BACKEND_MODEL, api_key="os.environ/OPENAI_API_KEY" ) result: Final = _poll_for_block(lambda: client.responses(scoped_key, model, "say hi", guardrails=[name])) diff --git a/tests/e2e/guardrails/test_streaming_guardrail_e2e.py b/tests/e2e/guardrails/test_streaming_guardrail_e2e.py index 911ddf9304b..324dd4107a4 100644 --- a/tests/e2e/guardrails/test_streaming_guardrail_e2e.py +++ b/tests/e2e/guardrails/test_streaming_guardrail_e2e.py @@ -22,6 +22,7 @@ import os import pytest from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Subject, meta from guardrails_client import ( BedrockGuardrailParamsBody, GuardrailsClient, @@ -39,6 +40,14 @@ class TestBedrockDuringCallStreaming: "guardrail.bedrock.during.blocks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_during_call_blocks_stream_before_first_chunk( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/guardrails/test_team_disable_global_guardrail_e2e.py b/tests/e2e/guardrails/test_team_disable_global_guardrail_e2e.py index db917d6ede9..0cc6106abfd 100644 --- a/tests/e2e/guardrails/test_team_disable_global_guardrail_e2e.py +++ b/tests/e2e/guardrails/test_team_disable_global_guardrail_e2e.py @@ -14,6 +14,7 @@ import pytest from e2e_config import unique_marker from e2e_http import UnknownApiError, unwrap +from e2e_metadata import Domain, Mode, Provider, Subject, meta from guardrails_client import GuardrailsClient from lifecycle import ResourceManager @@ -59,6 +60,14 @@ class TestTeamDisableGlobalGuardrail: "guardrail.litellm_content_filter.pre_call.blocks", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_global_guardrail_blocks_key_without_team_opt_out( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -72,6 +81,14 @@ class TestTeamDisableGlobalGuardrail: "guardrail.litellm_content_filter.pre_call.allows", exercised_on=["chat_completions"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_team_with_disable_flag_bypasses_global_guardrail( self, client: GuardrailsClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/guardrails/test_tool_permission_guardrail_e2e.py b/tests/e2e/guardrails/test_tool_permission_guardrail_e2e.py index 8d1047e53c7..d5a6a9d9866 100644 --- a/tests/e2e/guardrails/test_tool_permission_guardrail_e2e.py +++ b/tests/e2e/guardrails/test_tool_permission_guardrail_e2e.py @@ -25,6 +25,7 @@ import pytest from e2e_config import unique_marker from e2e_http import StreamingResponse, UnknownApiError +from e2e_metadata import Capability, Domain, Mode, Provider, Subject, meta from guardrails_client import ( GuardrailsClient, ToolPermissionParamsBody, @@ -101,6 +102,15 @@ def _tool_call_names(response: ChatResponse) -> tuple[str, ...]: class TestToolPermissionPreCall: @pytest.mark.covers("guardrail.tool_permission.pre_call.blocks", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_pre_call_blocks_tool_outside_the_allow_list( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -135,6 +145,15 @@ class TestToolPermissionPreCall: pytest.fail(f"tool_permission let a tool outside the allow-list through; got {result}") @pytest.mark.covers("guardrail.tool_permission.pre_call.allows", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.GUARDRAILS, + providers=(Provider.GEMINI,), + models=(MODEL,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_pre_call_allows_permitted_tool( self, client: GuardrailsClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/logging/datadog_reader.py b/tests/e2e/logging/datadog_reader.py index 368c20cb6aa..61b1b5a1536 100644 --- a/tests/e2e/logging/datadog_reader.py +++ b/tests/e2e/logging/datadog_reader.py @@ -32,6 +32,7 @@ from e2e_config import ( POLL_TIMEOUT, ) from e2e_http import URL, Headers, StreamingResponse, send +from e2e_metadata import step type SearchCall = Callable[[str, float], StreamingResponse] @@ -115,6 +116,7 @@ class DdLogsReader: sleep: Callable[[float], None] = field(default=time.sleep, repr=False) jitter: Callable[[], float] = field(default=random.random, repr=False) + @step("Search DataDog for logs carrying the marker {marker}") def events_for_marker(self, marker: str) -> list[DdLogEvent]: """Every ingested event whose attributes carry the marker. DataDog consumes the shipped JSON message into ``attributes`` and leaves the @@ -124,6 +126,7 @@ class DdLogsReader: it).""" return self.events_for_query(f"*:*{marker}*") + @step("Search DataDog for logs matching {query}") def events_for_query(self, query: str) -> list[DdLogEvent]: """Every ingested event the search query matches (failure payloads carry no prompt to mark, so failure scenarios query indexed attributes @@ -156,10 +159,12 @@ class DdLogsReader: timeout=timeout, ) + @step("Wait for DataDog to ingest logs carrying the marker {marker}, then watch for duplicates") def poll_events_for_marker(self, marker: str) -> list[DdLogEvent]: """``poll_events_for_query`` over the every-attribute marker scan.""" return self.poll_events_for_query(f"*:*{marker}*") + @step("Wait for DataDog to ingest logs matching {query}, then watch for duplicates") def poll_events_for_query(self, query: str) -> list[DdLogEvent]: """Poll until at least one matching event is searchable (the callback flushes in periodic batches and DataDog ingestion adds seconds of lag), diff --git a/tests/e2e/logging/gcs_reader.py b/tests/e2e/logging/gcs_reader.py index 60622c121ac..8abde23e221 100644 --- a/tests/e2e/logging/gcs_reader.py +++ b/tests/e2e/logging/gcs_reader.py @@ -30,6 +30,7 @@ from pydantic import BaseModel, ConfigDict, Field from e2e_config import POLL_INTERVAL, POLL_TIMEOUT from e2e_http import URL, Headers, probe +from e2e_metadata import step _GCS_API = "https://storage.googleapis.com" #: Tolerance for clock skew between this host and GCS object timestamps. @@ -44,7 +45,7 @@ class _ServiceAccount(BaseModel): model_config = ConfigDict(extra="ignore") client_email: str - private_key: str + private_key: str = Field(repr=False) class _GcsAuthHeaders(Headers): @@ -146,6 +147,7 @@ class GcsLogReader: ) return result.body + @step("Read the GCS bucket's log objects for the response {response_id}") def records_for_response_id(self, response_id: str, *, since: datetime) -> list[GcsLogRecord]: """Every payload written for ``response_id``: the direct ``{date}/{response_id}`` object plus any hit inside batch NDJSON @@ -168,6 +170,7 @@ class GcsLogReader: ) return records + @step("Wait for the log of the response {response_id} to land in the GCS bucket, then watch for duplicates") def poll_records_for_response_id(self, response_id: str, *, since: datetime) -> list[GcsLogRecord]: """Poll until the payload is readable (the gcs_bucket callback flushes on a ~20s timer), then keep re-reading for GCS_SETTLE_SECONDS - past a diff --git a/tests/e2e/logging/logging_client.py b/tests/e2e/logging/logging_client.py index c2f987ea33d..d522c01c054 100644 --- a/tests/e2e/logging/logging_client.py +++ b/tests/e2e/logging/logging_client.py @@ -24,6 +24,7 @@ import pytest from pydantic import BaseModel, ConfigDict, Field, JsonValue, TypeAdapter, ValidationError from e2e_config import POLL_INTERVAL, POLL_TIMEOUT, settle_propagation +from e2e_metadata import step from proxy_client import ProxyClient from e2e_http import ( URL, @@ -326,6 +327,7 @@ def observation_has_guardrail(obs: LangfuseObservation, *, guardrail_name: str) class LoggingClient: proxy: ProxyClient + @step("Generate a virtual key named {alias} with models: {models}") def key_with_alias( self, alias: str, @@ -347,9 +349,11 @@ class LoggingClient: ) ) + @step("Delete the virtual key") def delete_key(self, key: str) -> None: self.proxy.delete_key(key) + @step("Create the team {alias} with models: {models}") def create_team( self, alias: str, @@ -370,6 +374,7 @@ class LoggingClient: ) ).team_id + @step("Delete the team") def delete_team(self, team_id: str) -> None: _ = self.proxy.transport.post( "/team/delete", @@ -378,6 +383,7 @@ class LoggingClient: response_type=NoBody, ) + @step("Create the internal user {user_email}") def create_user(self, *, user_email: str, user_id: str | None = None) -> str: return unwrap( self.proxy.transport.post( @@ -392,6 +398,7 @@ class LoggingClient: ) ).user_id + @step("Delete the internal user") def delete_user(self, user_id: str) -> None: _ = self.proxy.transport.post( "/user/delete", @@ -400,6 +407,7 @@ class LoggingClient: response_type=NoBody, ) + @step("Create the organization {alias} with models: {models}") def create_org(self, alias: str, *, models: list[str]) -> str: return unwrap( self.proxy.transport.post( @@ -410,6 +418,7 @@ class LoggingClient: ) ).organization_id + @step("Delete the organization") def delete_org(self, organization_id: str) -> None: _ = self.proxy.transport.delete( "/organization/delete", @@ -418,6 +427,7 @@ class LoggingClient: response_type=NoBody, ) + @step("Add a Langfuse OTel logging callback for {callback_type} events to the team") def add_team_langfuse_callback( self, team_id: str, @@ -441,6 +451,7 @@ class LoggingClient: f"POST /team/{team_id}/callback must return status=success; got {response.status!r}" ) + @step("Create the tool_permission guardrail {name} that allows only the tool {allowed_tool}") def create_tool_permission_guardrail(self, name: str, *, allowed_tool: str) -> str: """Register a tool_permission guardrail that allows one tool and denies the rest.""" response = unwrap( @@ -474,6 +485,7 @@ class LoggingClient: settle_propagation(time.monotonic()) return guardrail_id + @step("Delete the guardrail") def delete_guardrail(self, guardrail_id: str) -> None: _ = self.proxy.transport.delete( f"/guardrails/{guardrail_id}", @@ -482,12 +494,15 @@ class LoggingClient: response_type=NoBody, ) + @step("Add a deployment named {model_name} that calls {litellm_params.model}") def create_model(self, model_name: str, litellm_params: LiteLLMParamsBody) -> str: return self.proxy.create_model(model_name, litellm_params) + @step("Delete the deployment") def delete_model(self, model_id: str) -> None: self.proxy.delete_model(model_id) + @step('Send a /chat/completions request to {model} with the prompt "{text}"') def chat(self, key: str, model: str, text: str) -> ChatResponse: return unwrap( self.proxy.chat( @@ -500,6 +515,7 @@ class LoggingClient: ) ) + @step('Send a /chat/completions request to {model} with stream={stream} and the prompt "{text}"') def chat_raw( self, key: str, @@ -529,6 +545,7 @@ class LoggingClient: json=body, ) + @step('Send a /v1/messages request to {model} with stream={stream} and the prompt "{text}"') def messages_raw( self, key: str, model: str, text: str, *, max_tokens: int = 16, stream: bool = False ) -> StreamingResponse: @@ -545,6 +562,7 @@ class LoggingClient: return self.proxy.transport.stream("/v1/messages", headers=self.proxy.transport.bearer(key), json=body) return self.proxy.transport.send("/v1/messages", headers=self.proxy.transport.bearer(key), json=body) + @step('Send a /v1/responses request to {model} with stream={stream} and the prompt "{text}"') def responses_raw( self, key: str, model: str, text: str, *, max_output_tokens: int = 64, stream: bool = False ) -> StreamingResponse: @@ -560,9 +578,11 @@ class LoggingClient: return self.proxy.transport.stream("/v1/responses", headers=self.proxy.transport.bearer(key), json=body) return self.proxy.transport.send("/v1/responses", headers=self.proxy.transport.bearer(key), json=body) + @step("Scrape the Prometheus metrics from /metrics") def scrape_metrics(self) -> str: return self.proxy.probe("/metrics", params=NoBody()).body + @step("Wait for the key's spend log in /spend/logs") def poll_proxy_spend_for_key( self, key: str, @@ -590,6 +610,7 @@ class LoggingClient: return row return None + @step("List observations from Langfuse") def list_langfuse_observations( self, creds: LangfuseCreds, @@ -616,6 +637,7 @@ class LoggingClient: case _: return [] + @step("Look up the Langfuse generation for the key {key_alias}") def find_langfuse_observation( self, creds: LangfuseCreds, @@ -634,6 +656,7 @@ class LoggingClient: return obs return None + @step("Wait for the Langfuse generation for the key {key_alias}") def poll_langfuse_observation( self, creds: LangfuseCreds, @@ -653,6 +676,7 @@ class LoggingClient: time.sleep(POLL_INTERVAL) return last + @step("Wait for the OTel v2 Langfuse generation for the key {key_alias}") def poll_langfuse_generation( self, creds: LangfuseCreds, *, key_alias: str, from_start_time: str ) -> LangfuseObservation | None: @@ -665,6 +689,7 @@ class LoggingClient: time.sleep(POLL_INTERVAL) return None + @step("Wait for the Langfuse trace of the key {key_alias} and every observation in it") def poll_langfuse_trace_observations( self, creds: LangfuseCreds, @@ -698,6 +723,7 @@ def build_logging_client(proxy: ProxyClient) -> LoggingClient: return LoggingClient(proxy=proxy) +@step("Read the proxy's callback list from /health/readiness/details") def readiness_details_body(client: LoggingClient) -> str: """/health/readiness/details, tolerating the 503 it serves while the ephemeral stack's DB leg blips: the recorded state the logging suites check diff --git a/tests/e2e/logging/s3_reader.py b/tests/e2e/logging/s3_reader.py index d605dec6096..572086c2695 100644 --- a/tests/e2e/logging/s3_reader.py +++ b/tests/e2e/logging/s3_reader.py @@ -24,6 +24,7 @@ import pytest from pydantic import BaseModel, ConfigDict from e2e_config import POLL_INTERVAL, POLL_TIMEOUT +from e2e_metadata import step if TYPE_CHECKING: from types_boto3_s3.client import S3Client @@ -53,17 +54,21 @@ class S3LogReader: bucket: str client: S3Client + @step("List the log objects in the S3 bucket under {prefix}") def list_keys(self, prefix: str) -> list[str]: response = self.client.list_objects_v2(Bucket=self.bucket, Prefix=prefix) return [obj["Key"] for obj in response.get("Contents", []) if "Key" in obj] + @step("Download a log object from the S3 bucket") def read_record(self, key: str) -> S3LogRecord: body = self.client.get_object(Bucket=self.bucket, Key=key)["Body"].read() return S3LogRecord.model_validate_json(body) + @step("Read the log objects in the S3 bucket under {prefix}") def records_matching(self, *, prefix: str, predicate: Callable[[S3LogRecord], bool]) -> list[S3LogRecord]: return [record for record in map(self.read_record, self.list_keys(prefix)) if predicate(record)] + @step("Wait for the request's log object to land in the S3 bucket under {prefix}, then watch for duplicates") def poll_records(self, *, prefix: str, predicate: Callable[[S3LogRecord], bool]) -> list[S3LogRecord]: """Poll until at least one matching object is listed (the s3_v2 callback flushes on a ~10s timer), then keep re-reading for diff --git a/tests/e2e/logging/test_datadog_log_e2e.py b/tests/e2e/logging/test_datadog_log_e2e.py index 2d57181c4bf..108985a3953 100644 --- a/tests/e2e/logging/test_datadog_log_e2e.py +++ b/tests/e2e/logging/test_datadog_log_e2e.py @@ -20,10 +20,12 @@ from __future__ import annotations import math import time +from typing import Final import pytest from datadog_reader import DdLogEvent, DdLogsReader from e2e_config import CHEAP_ANTHROPIC_MODEL, CHEAP_OPENAI_MODEL, unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from logging_client import INVALID_UPSTREAM_API_KEY, LoggingClient, first_ok, readiness_details_body from models import ChatMessage, LiteLLMParamsBody, ReliabilityChatBody, RouterSettingsOverride @@ -33,6 +35,7 @@ pytestmark = pytest.mark.e2e #: The active DataDog callback's name in /health/readiness/details success_callbacks. DD_LOGGER_NAME = "DataDogLogger" +FAILING_BACKEND_MODEL: Final = "anthropic/claude-haiku-4-5" class _DdMessagePayload(BaseModel): @@ -109,6 +112,15 @@ def _assert_exactly_one_event( class TestDataDogLogDelivery: @pytest.mark.covers("logging.datadog.success.exports_metric", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completions_emits_one_log_event( self, client: LoggingClient, dd_logs: DdLogsReader, resources: ResourceManager ) -> None: @@ -134,6 +146,15 @@ class TestDataDogLogDelivery: ) @pytest.mark.covers("logging.datadog.success.exports_metric", exercised_on=["messages"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_messages_emits_one_log_event( self, client: LoggingClient, dd_logs: DdLogsReader, resources: ResourceManager ) -> None: @@ -161,6 +182,15 @@ class TestDataDogLogDelivery: ) @pytest.mark.covers("logging.datadog.success.exports_metric", exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_emits_one_log_event( self, client: LoggingClient, dd_logs: DdLogsReader, resources: ResourceManager ) -> None: @@ -186,6 +216,15 @@ class TestDataDogLogDelivery: ) @pytest.mark.covers("logging.datadog.stream.exports_metric", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.STREAM, + ) + ) def test_chat_completions_stream_emits_one_log_event( self, client: LoggingClient, dd_logs: DdLogsReader, resources: ResourceManager ) -> None: @@ -232,6 +271,15 @@ class TestDataDogLogDelivery: ) @pytest.mark.covers("logging.datadog.stream.exports_metric", exercised_on=["messages"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.STREAM, + ) + ) def test_messages_stream_emits_one_log_event( self, client: LoggingClient, dd_logs: DdLogsReader, resources: ResourceManager ) -> None: @@ -276,6 +324,15 @@ class TestDataDogLogDelivery: ) @pytest.mark.covers("logging.datadog.stream.exports_metric", exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.STREAM, + ) + ) def test_responses_stream_emits_one_log_event( self, client: LoggingClient, dd_logs: DdLogsReader, resources: ResourceManager ) -> None: @@ -347,6 +404,15 @@ def _assert_exactly_one_failure_event(events: list[DdLogEvent], *, model_group: class TestDataDogFailureDelivery: @pytest.mark.covers("logging.datadog.failure.exports_metric", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(FAILING_BACKEND_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_failed_chat_completions_emits_one_error_event( self, client: LoggingClient, dd_logs: DdLogsReader, resources: ResourceManager ) -> None: @@ -366,7 +432,7 @@ class TestDataDogFailureDelivery: model_name = f"dd-err-{unique_marker()}" model_id = client.create_model( model_name, - LiteLLMParamsBody(model="anthropic/claude-haiku-4-5", api_key=INVALID_UPSTREAM_API_KEY), + LiteLLMParamsBody(model=FAILING_BACKEND_MODEL, api_key=INVALID_UPSTREAM_API_KEY), ) resources.defer(lambda: client.delete_model(model_id)) key = client.key_with_alias(f"dd-err-key-{unique_marker()}", models=[model_name]) @@ -399,6 +465,15 @@ class TestDataDogFailureDelivery: ) @pytest.mark.covers("logging.datadog.stream_failure.exports_metric", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(FAILING_BACKEND_MODEL,), + mode=Mode.STREAM, + ) + ) def test_failed_chat_completions_stream_emits_one_error_event( self, client: LoggingClient, dd_logs: DdLogsReader, resources: ResourceManager ) -> None: @@ -420,7 +495,7 @@ class TestDataDogFailureDelivery: model_id = client.create_model( model_name, LiteLLMParamsBody( - model="anthropic/claude-haiku-4-5", + model=FAILING_BACKEND_MODEL, api_key=INVALID_UPSTREAM_API_KEY, api_base="http://localhost:1", ), diff --git a/tests/e2e/logging/test_gcs_log_e2e.py b/tests/e2e/logging/test_gcs_log_e2e.py index 17ad1507049..259735e6cad 100644 --- a/tests/e2e/logging/test_gcs_log_e2e.py +++ b/tests/e2e/logging/test_gcs_log_e2e.py @@ -22,6 +22,7 @@ import math import pytest from e2e_config import CHEAP_ANTHROPIC_MODEL, unique_marker +from e2e_metadata import Domain, Mode, Provider, Subject, meta from gcs_reader import GcsLogReader, build_gcs_reader, utc_now from lifecycle import ResourceManager from logging_client import LoggingClient, completion_response_id, first_ok, readiness_details_body @@ -52,6 +53,14 @@ def _assert_gcs_configured(client: LoggingClient) -> None: class TestGcsLogDelivery: @pytest.mark.covers("logging.gcs_bucket.success.writes_object", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completions_writes_one_success_record( self, client: LoggingClient, gcs_logs: GcsLogReader, resources: ResourceManager ) -> None: diff --git a/tests/e2e/logging/test_langsmith_batch_serialization_e2e.py b/tests/e2e/logging/test_langsmith_batch_serialization_e2e.py index 874b2b6a045..e2dadf9e696 100644 --- a/tests/e2e/logging/test_langsmith_batch_serialization_e2e.py +++ b/tests/e2e/logging/test_langsmith_batch_serialization_e2e.py @@ -21,6 +21,7 @@ from typing import Final import pytest from e2e_config import CHEAP_OPENAI_MODEL, POLL_INTERVAL, POLL_TIMEOUT, unique_marker from e2e_http import Headers, Success, get_external +from e2e_metadata import Domain, Mode, Provider, Subject, meta from pydantic import BaseModel, ConfigDict, Field, JsonValue import litellm @@ -90,6 +91,14 @@ def _poll_run(creds: LangsmithCreds, run_id: uuid.UUID) -> LangsmithRun: class TestLangsmithBatchSerialization: @pytest.mark.asyncio @pytest.mark.covers("logging.langsmith.success.serializes_non_native_metadata") + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) async def test_non_json_native_metadata_reaches_langsmith(self) -> None: creds: Final = load_langsmith_creds() logger: Final = LangsmithLogger( diff --git a/tests/e2e/logging/test_otel_trace_e2e.py b/tests/e2e/logging/test_otel_trace_e2e.py index 4902b0703c3..01eaf5ce80c 100644 --- a/tests/e2e/logging/test_otel_trace_e2e.py +++ b/tests/e2e/logging/test_otel_trace_e2e.py @@ -25,6 +25,7 @@ from typing import Final import pytest from e2e_config import CHEAP_ANTHROPIC_MODEL, CHEAP_OPENAI_MODEL, OTEL_EXPORTER_ENDPOINT, unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from logging_client import INVALID_UPSTREAM_API_KEY, LoggingClient, first_ok, readiness_details_body from models import LiteLLMParamsBody @@ -34,6 +35,7 @@ from pydantic import BaseModel, ConfigDict, ValidationError pytestmark = pytest.mark.e2e MODEL = CHEAP_ANTHROPIC_MODEL +FAILING_BACKEND_MODEL: Final = "anthropic/claude-haiku-4-5" DB_SPAN_PREFIX = "postgres." #: The active OTEL v2 logger's name in /health/readiness/details success_callbacks. OTEL_V2_LOGGER_NAME = "OpenTelemetryV2" @@ -279,6 +281,15 @@ def _assert_error_span_contract(span: JaegerSpan) -> None: class TestOtelTraceCompleteness: @pytest.mark.covers("logging.otel.success.exports_metric", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completions_exports_complete_trace( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -312,6 +323,14 @@ class TestOtelTraceCompleteness: _assert_complete_trace(traces, route=route, genai_span=f"chat {MODEL}") @pytest.mark.covers("logging.otel.success.exports_metric", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) @pytest.mark.otel_tls def test_otel_export_over_tls_with_internal_ca_reaches_destination( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager @@ -337,6 +356,15 @@ class TestOtelTraceCompleteness: _assert_complete_trace(hits, route=route, genai_span=f"chat {MODEL}") @pytest.mark.covers("logging.otel.success.exports_metric", exercised_on=["messages"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_messages_exports_complete_trace( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -366,6 +394,15 @@ class TestOtelTraceCompleteness: _assert_complete_trace(traces, route=route, genai_span=f"chat {MODEL}") @pytest.mark.covers("logging.otel.success.exports_metric", exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_exports_complete_trace( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -396,6 +433,15 @@ class TestOtelTraceCompleteness: _assert_complete_trace(traces, route=route, genai_span=genai_span) @pytest.mark.covers("logging.otel.stream.exports_metric", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_chat_completions_stream_exports_complete_trace( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -444,6 +490,15 @@ class TestOtelTraceCompleteness: ) @pytest.mark.covers("logging.otel.stream.exports_metric", exercised_on=["messages"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_messages_stream_exports_complete_trace( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -492,6 +547,15 @@ class TestOtelTraceCompleteness: ) @pytest.mark.covers("logging.otel.stream.exports_metric", exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.STREAM, + ) + ) def test_responses_stream_exports_complete_trace( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -544,6 +608,15 @@ class TestOtelTraceCompleteness: ) @pytest.mark.covers("logging.otel.stream.records_ttft", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_chat_completions_stream_records_real_ttft( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -582,6 +655,15 @@ class TestOtelTraceCompleteness: _assert_real_ttft(traces.hits, genai_span=genai_span) @pytest.mark.covers("logging.otel.stream.records_ttft", exercised_on=["messages"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(MODEL,), + mode=Mode.STREAM, + ) + ) def test_messages_stream_records_real_ttft( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -620,6 +702,15 @@ class TestOtelTraceCompleteness: _assert_real_ttft(traces.hits, genai_span=genai_span) @pytest.mark.covers("logging.otel.stream.records_ttft", exercised_on=["responses"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL,), + mode=Mode.STREAM, + ) + ) def test_responses_stream_records_real_ttft( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -658,6 +749,15 @@ class TestOtelTraceCompleteness: _assert_real_ttft(traces.hits, genai_span=genai_span) @pytest.mark.covers("logging.otel.failure.exports_metric", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(FAILING_BACKEND_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_failed_chat_completions_error_span_attributes( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -677,7 +777,7 @@ class TestOtelTraceCompleteness: model_name = f"otel-err-{unique_marker()}" model_id = client.create_model( model_name, - LiteLLMParamsBody(model="anthropic/claude-haiku-4-5", api_key=INVALID_UPSTREAM_API_KEY), + LiteLLMParamsBody(model=FAILING_BACKEND_MODEL, api_key=INVALID_UPSTREAM_API_KEY), ) resources.defer(lambda: client.delete_model(model_id)) key = client.key_with_alias(f"otel-err-{unique_marker()}", models=[model_name]) @@ -711,6 +811,15 @@ class TestOtelTraceCompleteness: _assert_error_span_contract(genai) @pytest.mark.covers("logging.otel.failure.exports_metric", exercised_on=["messages"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(FAILING_BACKEND_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_failed_messages_error_span_attributes( self, client: LoggingClient, otel_reader: OtelReader, resources: ResourceManager ) -> None: @@ -729,7 +838,7 @@ class TestOtelTraceCompleteness: model_name = f"otel-err-{unique_marker()}" model_id = client.create_model( model_name, - LiteLLMParamsBody(model="anthropic/claude-haiku-4-5", api_key=INVALID_UPSTREAM_API_KEY), + LiteLLMParamsBody(model=FAILING_BACKEND_MODEL, api_key=INVALID_UPSTREAM_API_KEY), ) resources.defer(lambda: client.delete_model(model_id)) key = client.key_with_alias(f"otel-err-{unique_marker()}", models=[model_name]) diff --git a/tests/e2e/logging/test_otel_v2_langfuse_generation_output_e2e.py b/tests/e2e/logging/test_otel_v2_langfuse_generation_output_e2e.py index 9e08cf21da0..852e10636d0 100644 --- a/tests/e2e/logging/test_otel_v2_langfuse_generation_output_e2e.py +++ b/tests/e2e/logging/test_otel_v2_langfuse_generation_output_e2e.py @@ -24,6 +24,7 @@ from typing import Final import pytest from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from logging_client import LangfuseCreds, LangfuseObservation, LoggingClient, load_langfuse_creds from models import ( @@ -69,6 +70,13 @@ RED_SQUARE_PNG: Final = base64.b64decode( ) BOUNDED_OUTPUT_CHARS: Final = 1024 PLACEHOLDER_INPUT: Final = "default-message-value" +COMPLETION_BACKEND: Final = "openai/gpt-3.5-turbo-instruct" +IMAGE_BACKEND: Final = "openai/gpt-image-1-mini" +SPEECH_BACKEND: Final = "openai/gpt-4o-mini-tts" +TRANSCRIPTION_BACKEND: Final = "openai/gpt-4o-mini-transcribe" +MODERATION_BACKEND: Final = "openai/omni-moderation-latest" +MISTRAL_OCR_BACKEND: Final = "mistral/mistral-ocr-latest" +RERANK_BACKEND: Final = "cohere/rerank-v4.0-fast" class _OutputMessage(BaseModel): @@ -142,7 +150,7 @@ def _openai(model: str) -> LiteLLMParamsBody: def _mistral_ocr() -> LiteLLMParamsBody: - return LiteLLMParamsBody(model="mistral/mistral-ocr-latest", api_key="os.environ/MISTRAL_API_KEY") + return LiteLLMParamsBody(model=MISTRAL_OCR_BACKEND, api_key="os.environ/MISTRAL_API_KEY") def _langfuse_search_tool( @@ -171,10 +179,19 @@ def _langfuse_search_tool( class TestOtelV2LangfuseGenerationOutput: @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.COMPLETIONS, + providers=(Provider.OPENAI,), + models=(COMPLETION_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_completions_output_is_the_completion_text( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: - model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai("openai/gpt-3.5-turbo-instruct")) + model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai(COMPLETION_BACKEND)) started: Final = datetime.now(timezone.utc) response: Final = unwrap( client.proxy.transport.post( @@ -193,10 +210,19 @@ class TestOtelV2LangfuseGenerationOutput: ) @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["images_generations"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(IMAGE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_images_output_is_a_bounded_summary_without_base64( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: - model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai("openai/gpt-image-1-mini")) + model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai(IMAGE_BACKEND)) started: Final = datetime.now(timezone.utc) response: Final = unwrap( client.proxy.transport.post( @@ -220,10 +246,19 @@ class TestOtelV2LangfuseGenerationOutput: ) @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["audio_speech"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(SPEECH_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_speech_output_is_a_bounded_summary_without_audio_bytes( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: - model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai("openai/gpt-4o-mini-tts")) + model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai(SPEECH_BACKEND)) started: Final = datetime.now(timezone.utc) audio: Final = client.proxy.transport.stream_binary( "/v1/audio/speech", @@ -239,10 +274,19 @@ class TestOtelV2LangfuseGenerationOutput: ) @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["audio_transcriptions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.AUDIO, + providers=(Provider.OPENAI,), + models=(TRANSCRIPTION_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_transcription_output_is_the_transcript( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: - model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai("openai/gpt-4o-mini-transcribe")) + model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai(TRANSCRIPTION_BACKEND)) started: Final = datetime.now(timezone.utc) response: Final = unwrap( client.proxy.transport.upload( @@ -262,10 +306,19 @@ class TestOtelV2LangfuseGenerationOutput: assert transcript in output, f"generation output lacks the transcript {transcript!r}: {output!r}" @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["moderations"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.MODERATIONS, + providers=(Provider.OPENAI,), + models=(MODERATION_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_moderations_output_is_the_verdict( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: - model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai("openai/omni-moderation-latest")) + model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai(MODERATION_BACKEND)) started: Final = datetime.now(timezone.utc) response: Final = unwrap( client.proxy.transport.post( @@ -284,6 +337,15 @@ class TestOtelV2LangfuseGenerationOutput: ) @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["rerank"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.RERANK, + providers=(Provider.COHERE,), + models=(RERANK_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_rerank_output_is_the_ranked_indices_and_scores( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: @@ -291,7 +353,7 @@ class TestOtelV2LangfuseGenerationOutput: client, langfuse_creds, resources, - LiteLLMParamsBody(model="cohere/rerank-v4.0-fast", api_key="os.environ/COHERE_API_KEY"), + LiteLLMParamsBody(model=RERANK_BACKEND, api_key="os.environ/COHERE_API_KEY"), ) query: Final = f"What is the capital of France? {unique_marker()}" started: Final = datetime.now(timezone.utc) @@ -317,6 +379,15 @@ class TestOtelV2LangfuseGenerationOutput: assert output == "\n\n".join(ranked), f"generation output is not the ranked indices and scores: {output!r}" @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["ocr"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.OCR, + providers=(Provider.MISTRAL,), + models=(MISTRAL_OCR_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_ocr_input_is_the_document_url( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: @@ -334,6 +405,15 @@ class TestOtelV2LangfuseGenerationOutput: assert response.pages[0].markdown in _output_text(generation) @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["ocr"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.OCR, + providers=(Provider.MISTRAL,), + models=(MISTRAL_OCR_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_ocr_upload_input_is_a_bounded_document_summary_without_base64( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: @@ -360,10 +440,19 @@ class TestOtelV2LangfuseGenerationOutput: ) @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["images_edits"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.IMAGES, + providers=(Provider.OPENAI,), + models=(IMAGE_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_image_edit_input_is_the_edit_prompt( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: - model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai("openai/gpt-image-1-mini")) + model, key, alias = _langfuse_key(client, langfuse_creds, resources, _openai(IMAGE_BACKEND)) prompt: Final = f"make the square blue {unique_marker()}" started: Final = datetime.now(timezone.utc) response: Final = unwrap( @@ -388,6 +477,11 @@ class TestOtelV2LangfuseGenerationOutput: assert _output_text(generation).startswith("b64_json image (") @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["search"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + ) + ) def test_search_input_is_the_query_and_output_the_results( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: diff --git a/tests/e2e/logging/test_prometheus_cardinality_e2e.py b/tests/e2e/logging/test_prometheus_cardinality_e2e.py index e2d164d8b4c..597e1e814e3 100644 --- a/tests/e2e/logging/test_prometheus_cardinality_e2e.py +++ b/tests/e2e/logging/test_prometheus_cardinality_e2e.py @@ -24,6 +24,7 @@ import pytest from prometheus_client.parser import text_string_to_metric_families from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from logging_client import LoggingClient @@ -47,6 +48,15 @@ def _aliases_in_metric(exposition: str, metric: str, label: str) -> frozenset[st class TestPrometheusPerKeyCardinality: @pytest.mark.covers("logging.prometheus.success.exports_metric", exercised_on=[]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.METRICS, + providers=(Provider.GEMINI,), + models=(DRIVER_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_distinct_key_aliases_produce_distinct_series( self, client: LoggingClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/logging/test_prometheus_queue_time_e2e.py b/tests/e2e/logging/test_prometheus_queue_time_e2e.py index 1f3c111bb65..9668322b362 100644 --- a/tests/e2e/logging/test_prometheus_queue_time_e2e.py +++ b/tests/e2e/logging/test_prometheus_queue_time_e2e.py @@ -6,6 +6,7 @@ import pytest from prometheus_client.parser import text_string_to_metric_families from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from logging_client import LoggingClient @@ -30,6 +31,15 @@ def _observation_count(exposition: str, alias: str) -> float | None: class TestPrometheusRequestQueueTime: @pytest.mark.covers("logging.prometheus.success.records_queue_time") + @meta( + Subject( + domain=Domain.OBSERVABILITY, + route=Route.METRICS, + providers=(Provider.GEMINI,), + models=(DRIVER_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_queue_time_histogram_records_an_observation( self, client: LoggingClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/logging/test_s3_log_e2e.py b/tests/e2e/logging/test_s3_log_e2e.py index 1612fa315a6..e917276edb0 100644 --- a/tests/e2e/logging/test_s3_log_e2e.py +++ b/tests/e2e/logging/test_s3_log_e2e.py @@ -24,10 +24,12 @@ from __future__ import annotations import math import re import time +from typing import Final import pytest from e2e_config import CHEAP_ANTHROPIC_MODEL, S3_PARTITION_GRANULARITY, unique_marker +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from logging_client import ( INVALID_UPSTREAM_API_KEY, @@ -43,6 +45,7 @@ pytestmark = pytest.mark.e2e #: The active s3_v2 callback's name in /health/readiness/details success_callbacks. S3_LOGGER_NAME = "S3Logger" +UNREACHABLE_ANTHROPIC_BACKEND: Final = "anthropic/claude-haiku-4-5" @pytest.fixture(scope="session") @@ -64,6 +67,14 @@ def _assert_s3_configured(client: LoggingClient) -> None: class TestS3LogDelivery: @pytest.mark.covers("logging.s3.success.writes_object", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completions_writes_one_success_object( self, client: LoggingClient, s3_logs: S3LogReader, resources: ResourceManager ) -> None: @@ -108,6 +119,14 @@ class TestS3LogDelivery: ), f"payload response_cost {record.response_cost!r} must equal the header cost {outcome.response_cost}" @pytest.mark.covers("logging.s3.success.partition_layout", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completions_object_key_follows_the_partition_granularity( self, client: LoggingClient, s3_logs: S3LogReader, resources: ResourceManager ) -> None: @@ -148,6 +167,14 @@ class TestS3LogDelivery: ) @pytest.mark.covers("logging.s3.failure.writes_object", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.ANTHROPIC,), + models=(UNREACHABLE_ANTHROPIC_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completions_failure_writes_one_object( self, client: LoggingClient, s3_logs: S3LogReader, resources: ResourceManager ) -> None: @@ -167,7 +194,7 @@ class TestS3LogDelivery: model_name = f"s3-err-{unique_marker()}" model_id = client.create_model( model_name, - LiteLLMParamsBody(model="anthropic/claude-haiku-4-5", api_key=INVALID_UPSTREAM_API_KEY), + LiteLLMParamsBody(model=UNREACHABLE_ANTHROPIC_BACKEND, api_key=INVALID_UPSTREAM_API_KEY), ) resources.defer(lambda: client.delete_model(model_id)) alias = f"s3-err-key-{unique_marker()}" diff --git a/tests/e2e/logging/test_team_langfuse_callback_e2e.py b/tests/e2e/logging/test_team_langfuse_callback_e2e.py index 89cd45c9f16..f2ece7f5d90 100644 --- a/tests/e2e/logging/test_team_langfuse_callback_e2e.py +++ b/tests/e2e/logging/test_team_langfuse_callback_e2e.py @@ -19,6 +19,7 @@ import time import pytest from e2e_config import CHEAP_ANTHROPIC_MODEL, unique_marker +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from logging_client import ( LangfuseCreds, @@ -46,6 +47,14 @@ def langfuse_creds() -> LangfuseCreds: class TestTeamLangfuseCallback: @pytest.mark.covers("logging.langfuse.success.logs_spend", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_team_callback_delivers_and_isolates( self, client: LoggingClient, langfuse_creds: LangfuseCreds, resources: ResourceManager ) -> None: diff --git a/tests/e2e/logging/test_weave_log_e2e.py b/tests/e2e/logging/test_weave_log_e2e.py index dab5993c87a..edd8ec7c3c3 100644 --- a/tests/e2e/logging/test_weave_log_e2e.py +++ b/tests/e2e/logging/test_weave_log_e2e.py @@ -21,11 +21,13 @@ query API; nothing is mocked. from __future__ import annotations import time +from typing import Final import pytest from e2e_config import CHEAP_ANTHROPIC_MODEL, unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from logging_client import ( INVALID_UPSTREAM_API_KEY, @@ -40,6 +42,8 @@ from weave_reader import WeaveCall, WeaveReader, build_weave_reader pytestmark = pytest.mark.e2e +UNREACHABLE_ANTHROPIC_BACKEND: Final = "anthropic/claude-haiku-4-5" + @pytest.fixture(scope="session") def weave_creds() -> WeaveCreds: @@ -79,6 +83,14 @@ WEAVE_STAGE_RED_REASON = ( class TestWeaveLogDelivery: @pytest.mark.skip(reason=WEAVE_STAGE_RED_REASON) @pytest.mark.covers("logging.niche_integrations.success.logs_spend", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completions_delivers_one_call_with_spend( self, client: LoggingClient, @@ -121,6 +133,14 @@ class TestWeaveLogDelivery: @pytest.mark.skip(reason=WEAVE_STAGE_RED_REASON) @pytest.mark.covers("logging.niche_integrations.failure.logs_spend", exercised_on=["chat_completions"]) + @meta( + Subject( + domain=Domain.OBSERVABILITY, + providers=(Provider.ANTHROPIC,), + models=(UNREACHABLE_ANTHROPIC_BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_failed_chat_completions_delivers_one_error_call( self, client: LoggingClient, @@ -135,7 +155,7 @@ class TestWeaveLogDelivery: model_name = f"weave-err-{unique_marker()}" model_id = client.create_model( model_name, - LiteLLMParamsBody(model="anthropic/claude-haiku-4-5", api_key=INVALID_UPSTREAM_API_KEY), + LiteLLMParamsBody(model=UNREACHABLE_ANTHROPIC_BACKEND, api_key=INVALID_UPSTREAM_API_KEY), ) resources.defer(lambda: client.delete_model(model_id)) key = client.key_with_alias( diff --git a/tests/e2e/logging/weave_reader.py b/tests/e2e/logging/weave_reader.py index 2f8f759d299..c43e3280b39 100644 --- a/tests/e2e/logging/weave_reader.py +++ b/tests/e2e/logging/weave_reader.py @@ -28,7 +28,7 @@ import base64 import json import os import time -from dataclasses import dataclass +from dataclasses import dataclass, field from itertools import count, takewhile from typing import Final @@ -37,6 +37,7 @@ from pydantic import BaseModel, ConfigDict, Field from e2e_config import POLL_INTERVAL, POLL_TIMEOUT from e2e_http import URL, AuthHeaders, send +from e2e_metadata import step _WEAVE_TRACE_API: Final = "https://trace.wandb.ai" @@ -201,7 +202,7 @@ class WeaveCall(BaseModel): @dataclass(frozen=True, slots=True) class WeaveReader: project_id: str - api_key: str + api_key: str = field(repr=False) @property def _headers(self) -> AuthHeaders: @@ -232,6 +233,7 @@ class WeaveReader: ) return tuple(WeaveCall.model_validate_json(line) for line in outcome.body.splitlines() if line.strip()) + @step("Read the Weave {op} calls carrying the marker {marker}") def calls_matching(self, marker: str, *, since: float, op: str = LITELLM_REQUEST_OP) -> tuple[WeaveCall, ...]: """Every call under ``op`` started after ``since`` whose inputs carry ``marker``, paging until the window is exhausted. @@ -247,6 +249,7 @@ class WeaveReader: ) return tuple(call for page in pages for call in page if call.mentions(marker)) + @step("Wait for Weave to ingest a {op} call carrying the marker {marker}, then watch for duplicates") def poll_calls_matching(self, marker: str, *, since: float, op: str = LITELLM_REQUEST_OP) -> tuple[WeaveCall, ...]: """Poll until the call is readable, then keep re-reading for WEAVE_SETTLE_SECONDS so a duplicate exported by a later batch flush From 77fc3315e55b5dc22a99b4f3c0a7d76eb7e8c367 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 7 Oct 2026 17:32:39 +0000 Subject: [PATCH 06/13] fix(caching): skip the cache past max_messages and keep tool_result text in semantic prompts (#43878) * fix(caching): keep tool calls and tool results in semantic cache prompts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(caching): keep semantic tool prompt helpers within lint budgets Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(caching): keep structured function_call_output text in semantic prompts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(caching): split Responses text-field collection to stay within complexity budget Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(caching): tag each tool result with the position of the call it answers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(caching): encode tool result position and output together so tool text cannot forge result tags Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(caching): expect encoded tool result record in qdrant semantic prompt parity case Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(caching): cover tool result arrangements, SDK clients, concurrency and qdrant outage for semantic cache Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(caching): embed every semantic cache prompt field except volatile ones Replace the per-shape allowlist in the Python and Rust semantic cache prompt walkers with one include-by-default walker. Plain text keeps its old concatenation; any other block or message is embedded as compact JSON with call ids mapped to ordinals, cache_control dropped, and signatures, encrypted content and base64 data replaced with a short sha256 digest. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(code-quality): allow the bounded semantic cache prompt walkers in the recursion check Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(rust): expect structured JSON for unknown fields in redis and valkey semantic prompts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * Revert "test(rust): expect structured JSON for unknown fields in redis and valkey semantic prompts" This reverts commit 86c82b949b3da25461e72f317b0f54572ea5359f. * Revert "test(code-quality): allow the bounded semantic cache prompt walkers in the recursion check" This reverts commit 39efb5d9da8292844a9844e01d095702b11ecec4. * Revert "feat(caching): embed every semantic cache prompt field except volatile ones" This reverts commit 5aed3ab3de3597dec22d3afcb33db7196f0a2a24. * refactor(caching): rename get_str_from_messages_with_tools to get_semantic_cache_prompt_from_messages Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(caching): split semantic cache prompt extraction by API format Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(caching): drop TypeIs guard and register Responses prompt walker with the recursion check Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(caching): pick the Responses text field without a Final inside a loop Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(caching): walk semantic cache prompts as plain dicts, dumping pydantic items once up front Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(caching): write the semantic cache prompt builders as plain loops Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(caching): skip the cache past max_messages and keep tool_result text in semantic prompts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(caching): drop formatting-only churn from the redis semantic cache tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * ci(integration): drop the caching group wiring that main already carries Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(caching): recurse into tool_result content in the semantic cache prompt helper Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): check the max_messages cap on the shared exact-cache proxy Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(caching): read list-form function_call_output text in semantic cache prompts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(caching): extract nested Responses input lookup to keep walker under complexity limit Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * Revert "refactor(caching): extract nested Responses input lookup to keep walker under complexity limit" This reverts commit 0665296bf1d783583ffc7f494e111ef58fec2b18. * style(caching): suppress C901 on the Responses input walker instead of splitting it Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm-rust/crates/cache/src/semantic.rs | 37 ++++- litellm-rust/crates/cache/tests/semantic.rs | 34 +++++ litellm/caching/caching.py | 17 +++ litellm/caching/qdrant_semantic_cache.py | 10 +- litellm/caching/redis_semantic_cache.py | 11 +- .../prompt_templates/common_utils.py | 27 ++++ .../caching/test_cache_max_messages.py | 144 ++++++++++++++++++ tests/unit/caching/test_caching.py | 65 +++++++- .../unit/caching/test_redis_semantic_cache.py | 34 +++++ ...ore_utils_prompt_templates_common_utils.py | 92 +++++++++++ 10 files changed, 456 insertions(+), 15 deletions(-) create mode 100644 tests/integration/caching/test_cache_max_messages.py diff --git a/litellm-rust/crates/cache/src/semantic.rs b/litellm-rust/crates/cache/src/semantic.rs index c88706a213b..6200645e555 100644 --- a/litellm-rust/crates/cache/src/semantic.rs +++ b/litellm-rust/crates/cache/src/semantic.rs @@ -83,17 +83,16 @@ impl Embedder for PreparedEmbedding { } } -/// `get_str_from_messages`: every message's text content followed by its search results. +/// `get_semantic_cache_prompt_from_messages`: every message's text content, including the text of +/// Messages API `tool_result` blocks, followed by its search results. pub fn str_from_messages(messages: &[Value]) -> String { let mut text = String::new(); for message in messages.iter().filter_map(Value::as_object) { match message.get("content") { Some(Value::String(content)) => text.push_str(content), - Some(Value::Array(parts)) => { - for part in parts { - if let Some(part_text) = part.get("text").and_then(Value::as_str) { - text.push_str(part_text); - } + Some(Value::Array(blocks)) => { + for block in blocks { + push_block_text(&mut text, block); } } _ => {} @@ -103,6 +102,28 @@ pub fn str_from_messages(messages: &[Value]) -> String { text } +fn push_block_text(text: &mut String, block: &Value) { + if block.get("type").and_then(Value::as_str) != Some("tool_result") { + push_text_field(text, block); + return; + } + match block.get("content") { + Some(Value::String(result)) => text.push_str(result), + Some(Value::Array(blocks)) => { + for inner in blocks { + push_text_field(text, inner); + } + } + _ => {} + } +} + +fn push_text_field(text: &mut String, block: &Value) { + if let Some(block_text) = block.get("text").and_then(Value::as_str) { + text.push_str(block_text); + } +} + /// The messages prompt Qdrant embeds: `None` when the request carries no messages. pub fn prompt_from_messages(context: &SemanticCacheContext) -> Option { let messages = context.messages.as_ref()?.as_array()?; @@ -163,6 +184,10 @@ fn collect_input_text(value: &Value, parts: &mut Vec) { collect_input_text(content, parts); return; } + if let Some(output) = map.get("output").filter(|output| output.is_array()) { + collect_input_text(output, parts); + return; + } for key in ["text", "output", "input_text", "output_text"] { if let Some(Value::String(text)) = map.get(key) && push_trimmed(text, parts) diff --git a/litellm-rust/crates/cache/tests/semantic.rs b/litellm-rust/crates/cache/tests/semantic.rs index 97a552a8010..76af1863e1f 100644 --- a/litellm-rust/crates/cache/tests/semantic.rs +++ b/litellm-rust/crates/cache/tests/semantic.rs @@ -30,6 +30,31 @@ fn context(messages: Option, input: Option) -> SemanticCacheContex ]}]), "What is this?", )] +#[case::tool_result_string( + json!([ + {"role": "user", "content": "list the files"}, + {"role": "assistant", "content": [ + {"type": "tool_use", "id": "toolu_1", "name": "Bash", "input": {"command": "ls"}}, + ]}, + {"role": "user", "content": [ + {"type": "tool_result", "tool_use_id": "toolu_1", "content": "calc.py test_calc.py"}, + ]}, + ]), + "list the filescalc.py test_calc.py", +)] +#[case::tool_result_blocks( + json!([{"role": "user", "content": [ + {"type": "tool_result", "tool_use_id": "toolu_1", "content": [ + {"type": "text", "text": "x = 1"}, + {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": ""}}, + ]}, + ]}]), + "x = 1", +)] +#[case::tool_result_without_content( + json!([{"role": "user", "content": [{"type": "tool_result", "tool_use_id": "toolu_1"}]}]), + "", +)] #[case::missing_null_and_empty_content( json!([{"role": "assistant"}, {"role": "assistant", "content": null}, {"role": "user", "content": ""}]), "", @@ -166,6 +191,15 @@ fn prompt_from_messages_reads_messages_only( ])), Some("model dump prompt\ndict prompt\ninline prompt"), )] +#[case::function_call_output_blocks( + None, + Some(json!([ + {"role": "user", "content": "update the config"}, + {"type": "function_call", "call_id": "c1", "name": "write_file", "arguments": "{\"path\": \"a\"}"}, + {"type": "function_call_output", "call_id": "c1", "output": [{"type": "input_text", "text": "wrote a"}]}, + ])), + Some("update the config\nwrote a"), +)] #[case::object_content( None, Some(json!({"content": [{"text": "object content prompt"}]})), diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index 1dc4de04dc1..85ef5a93937 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -71,6 +71,17 @@ class CacheMode(str, Enum): #### LiteLLM.Completion / Embedding Cache #### +def _request_message_count(kwargs: Mapping[str, object]) -> int: + """Chat and Messages API `messages`, else Responses API `input` items; embedding `input` strings count as none""" + messages: Final = kwargs.get("messages") + if isinstance(messages, list): + return len(messages) + input_items: Final = kwargs.get("input") + if not isinstance(input_items, list): + return 0 + return sum(1 for item in input_items if isinstance(item, (Mapping, BaseModel))) + + class Cache: def __init__( self, @@ -119,6 +130,7 @@ class Cache: semantic_cache_embedding_max_input_tokens: int | None = None, semantic_cache_embedding_timeout: float | None = None, semantic_cache_scope: str = SemanticCacheScope.KEY.value, + max_messages: int | None = 4, # GCP IAM authentication parameters gcp_service_account: str | None = None, gcp_ssl_ca_certs: str | None = None, @@ -148,6 +160,7 @@ class Cache: semantic_cache_embedding_max_input_tokens (int, optional): Truncate prompts to this many tokens before embedding them for semantic caching. Defaults to the embedding deployment's configured max_input_tokens. semantic_cache_embedding_timeout (float, optional): Seconds a semantic-cache lookup may spend embedding the prompt before it gives up and lets the request continue to the LLM. Defaults to SEMANTIC_CACHE_EMBEDDING_TIMEOUT_SECONDS. semantic_cache_scope (str, optional): "key" isolates semantic-cache buckets per key/team/org. "end_user" additionally isolates per end user (falls back to the key scope when the request carries no end-user id). Defaults to "key". + max_messages (int, optional): Requests with more `messages` (or Responses API `input` items) than this are neither looked up nor stored, so long agent conversations never serve or create a cache entry. None disables the limit. Defaults to 4. # Disk Cache Args disk_cache_dir (str, optional): The directory for the disk cache. Defaults to None. @@ -298,6 +311,7 @@ class Cache: self.ttl = ttl self.mode: CacheMode = mode or CacheMode.default_on self.semantic_cache_scope: str = SemanticCacheScope(semantic_cache_scope).value + self.max_messages: int | None = max_messages if self.type == LiteLLMCacheType.LOCAL and default_in_memory_ttl is not None: self.ttl = default_in_memory_ttl @@ -933,7 +947,10 @@ class Cache: If cache is default_on then this is True If cache is default_off then this is only true when user has opted in to use cache + Always False once the request carries more than `max_messages` messages """ + if self.max_messages is not None and _request_message_count(kwargs) > self.max_messages: + return False if self.mode == CacheMode.default_on: return True diff --git a/litellm/caching/qdrant_semantic_cache.py b/litellm/caching/qdrant_semantic_cache.py index 8dac3f2eef9..eb88df066f6 100644 --- a/litellm/caching/qdrant_semantic_cache.py +++ b/litellm/caching/qdrant_semantic_cache.py @@ -24,7 +24,7 @@ from litellm.constants import ( ) from litellm.litellm_core_utils.asyncify import asyncify from litellm.litellm_core_utils.prompt_templates.common_utils import ( - get_str_from_messages, + get_semantic_cache_prompt_from_messages, ) from litellm.types.utils import EmbeddingResponse @@ -286,7 +286,7 @@ class QdrantSemanticCache(BaseCache): # get the prompt messages: Final = kwargs["messages"] - prompt: Final = get_str_from_messages(messages) + prompt: Final = get_semantic_cache_prompt_from_messages(messages) # create an embedding for prompt embedding_response: Final = cast( @@ -325,7 +325,7 @@ class QdrantSemanticCache(BaseCache): # get the messages messages: Final = kwargs["messages"] - prompt: Final = get_str_from_messages(messages) + prompt: Final = get_semantic_cache_prompt_from_messages(messages) # convert to embedding embedding_response: Final = cast( @@ -400,7 +400,7 @@ class QdrantSemanticCache(BaseCache): # get the prompt messages: Final = kwargs["messages"] - prompt: Final = get_str_from_messages(messages) + prompt: Final = get_semantic_cache_prompt_from_messages(messages) embedding_response: Final = await self._get_async_embedding(prompt, metadata=kwargs.get("metadata")) # get the embedding @@ -435,7 +435,7 @@ class QdrantSemanticCache(BaseCache): # get the messages messages: Final = kwargs["messages"] - prompt: Final = get_str_from_messages(messages) + prompt: Final = get_semantic_cache_prompt_from_messages(messages) embedding_response: Final = await self._get_async_embedding(prompt, metadata=kwargs.get("metadata")) diff --git a/litellm/caching/redis_semantic_cache.py b/litellm/caching/redis_semantic_cache.py index d4c815e15b7..8f99d76ba46 100644 --- a/litellm/caching/redis_semantic_cache.py +++ b/litellm/caching/redis_semantic_cache.py @@ -21,7 +21,7 @@ from litellm._logging import print_verbose, verbose_logger from litellm.constants import SEMANTIC_CACHE_EMBEDDING_TIMEOUT_SECONDS from litellm.litellm_core_utils.asyncify import asyncify from litellm.litellm_core_utils.prompt_templates.common_utils import ( - get_str_from_messages, + get_semantic_cache_prompt_from_messages, ) from litellm.types.utils import EmbeddingResponse @@ -263,7 +263,7 @@ class RedisSemanticCache(BaseCache): """ messages: Final = kwargs.get("messages") if messages: - return get_str_from_messages(messages) + return get_semantic_cache_prompt_from_messages(messages) if "input" not in kwargs: return None @@ -274,7 +274,7 @@ class RedisSemanticCache(BaseCache): return prompt or None @classmethod - def _collect_responses_input_text(cls, value: object, prompt_parts: list[str]) -> None: + def _collect_responses_input_text(cls, value: object, prompt_parts: list[str]) -> None: # noqa: C901 # one branch per Responses input shape value = cls._coerce_response_input_value(value) if value is None: return @@ -296,6 +296,11 @@ class RedisSemanticCache(BaseCache): cls._collect_responses_input_text(content, prompt_parts) return + output = value.get("output") + if isinstance(output, list): + cls._collect_responses_input_text(output, prompt_parts) + return + for text_key in ("text", "output", "input_text", "output_text"): text_value = value.get(text_key) if isinstance(text_value, str): diff --git a/litellm/litellm_core_utils/prompt_templates/common_utils.py b/litellm/litellm_core_utils/prompt_templates/common_utils.py index 49c4198c939..1f75125df09 100644 --- a/litellm/litellm_core_utils/prompt_templates/common_utils.py +++ b/litellm/litellm_core_utils/prompt_templates/common_utils.py @@ -192,6 +192,33 @@ def get_str_from_messages(messages: list[AllMessageValues]) -> str: return text +def get_semantic_cache_prompt_from_messages(messages: Sequence[Mapping[str, object]]) -> str: + """ + The text a semantic cache embeds for a request: `get_str_from_messages` plus the text inside + Messages API `tool_result` blocks, so a tool turn does not embed identically to the turn before it + """ + return "".join( + _semantic_cache_content_text(message.get("content")) + + extract_search_results_text(message.get("search_results")) + for message in messages + ) + + +def _semantic_cache_content_text(content: object) -> str: + if isinstance(content, str): + return content + if not isinstance(content, list): + return "" + return "".join(_semantic_cache_block_text(block) for block in content if isinstance(block, Mapping)) + + +def _semantic_cache_block_text(block: Mapping[str, object]) -> str: + if block.get("type") == "tool_result": + return _semantic_cache_content_text(block.get("content")) + text: Final = block.get("text") + return text if isinstance(text, str) else "" + + def is_non_content_values_set(message: AllMessageValues) -> bool: ignore_keys: Final = ["content", "role", "name"] return any(message.get(key, None) is not None for key in message if key not in ignore_keys) diff --git a/tests/integration/caching/test_cache_max_messages.py b/tests/integration/caching/test_cache_max_messages.py new file mode 100644 index 00000000000..89401b7c4ec --- /dev/null +++ b/tests/integration/caching/test_cache_max_messages.py @@ -0,0 +1,144 @@ +import json +import os +import uuid +from collections.abc import Callable +from typing import Final + +import pytest +from integration._support.anthropic_sse import message_json +from integration._support.client import Gateway, eventually, object_value, string_value +from integration._support.openai_wire import chat_reply, responses_reply +from integration._support.provider import SharedProvider +from integration._support.wire import Reply +from pydantic import JsonValue +from redis import Redis + +_Turns = Callable[[str], tuple[list[JsonValue], list[JsonValue]]] + + +def _claude_code_turns(task: str) -> tuple[list[JsonValue], list[JsonValue]]: + """Turns 1 and 3 of a Claude Code session on /v1/messages: 1 and 5 messages""" + first: Final[list[JsonValue]] = [{"role": "user", "content": task}] + third: Final[list[JsonValue]] = [ + *first, + { + "role": "assistant", + "content": [{"type": "tool_use", "id": "toolu_1", "name": "Bash", "input": {"command": "ls"}}], + }, + { + "role": "user", + "content": [{"type": "tool_result", "tool_use_id": "toolu_1", "content": "calc.py test_calc.py"}], + }, + { + "role": "assistant", + "content": [{"type": "tool_use", "id": "toolu_2", "name": "Read", "input": {"file_path": "calc.py"}}], + }, + { + "role": "user", + "content": [{"type": "tool_result", "tool_use_id": "toolu_2", "content": "def add(a, b): return a - b"}], + }, + ] + return first, third + + +def _agent_turns(task: str) -> tuple[list[JsonValue], list[JsonValue]]: + """Turns 1 and 3 of an OpenAI tool loop on /v1/chat/completions: 2 and 6 messages""" + + def call(call_id: str, path: str) -> list[JsonValue]: + function: Final[JsonValue] = {"name": "write_file", "arguments": json.dumps({"path": path})} + return [ + { + "role": "assistant", + "content": None, + "tool_calls": [{"id": call_id, "type": "function", "function": function}], + }, + {"role": "tool", "tool_call_id": call_id, "content": f"wrote {path}"}, + ] + + first: Final[list[JsonValue]] = [ + {"role": "system", "content": "You are a coding agent"}, + {"role": "user", "content": task}, + ] + return first, [*first, *call("call_1", "a.yaml"), *call("call_2", "b.yaml")] + + +def _responses_turns(task: str) -> tuple[list[JsonValue], list[JsonValue]]: + """Turns 1 and 3 of an agent on /v1/responses: 1 and 5 input items""" + + def call(call_id: str, path: str) -> list[JsonValue]: + return [ + { + "type": "function_call", + "call_id": call_id, + "name": "write_file", + "arguments": json.dumps({"path": path}), + }, + {"type": "function_call_output", "call_id": call_id, "output": f"wrote {path}"}, + ] + + first: Final[list[JsonValue]] = [{"role": "user", "content": task}] + return first, [*first, *call("call_1", "a.yaml"), *call("call_2", "b.yaml")] + + +def _body(path: str, model: str, conversation: list[JsonValue]) -> dict[str, JsonValue]: + if path == "/v1/responses": + return {"model": model, "input": conversation} + return {"model": model, "max_tokens": 16, "messages": conversation} + + +def _reply(path: str, text: str) -> Reply: + identity: Final = f"id_{uuid.uuid4().hex}" + if path == "/v1/messages": + return Reply(body=message_json(identity, "claude-sonnet-5-5", text)) + if path == "/v1/responses": + return responses_reply(identity, "gpt-5.6-sol", text, stream=False) + return chat_reply(identity, "gpt-5.4", text, stream=False) + + +def _answer(path: str, payload: dict[str, JsonValue]) -> str: + if path == "/v1/messages": + return string_value(_first(payload["content"])["text"]) + if path == "/v1/responses": + return string_value(_first(_first(payload["output"])["content"])["text"]) + return string_value(object_value(_first(payload["choices"])["message"])["content"]) + + +def _first(value: JsonValue) -> dict[str, JsonValue]: + assert isinstance(value, list), value + return object_value(value[0]) + + +def _cached_responses(redis: Redis) -> frozenset[bytes]: + digests: Final = tuple(key for key in redis.scan_iter() if len(key) == 64) + return frozenset(key for key in digests if b'"response"' in (redis.get(key) or b"")) + + +@pytest.mark.parametrize( + ("path", "model", "turns"), + [ + pytest.param("/v1/messages", "anthropic/claude-sonnet-5-5", _claude_code_turns, id="messages"), + pytest.param("/v1/chat/completions", "openai/gpt-5.4", _agent_turns, id="chat-completions"), + pytest.param("/v1/responses", "openai/responses/gpt-5.6-sol", _responses_turns, id="responses"), + ], +) +def test_cache_serves_a_turn_under_max_messages_and_skips_one_past_it( + gateway: Gateway, provider: SharedProvider, path: str, model: str, turns: _Turns +) -> None: + under_cap, past_cap = turns(f"update the config {uuid.uuid4().hex}") + provider.expect(_reply(path, "first answer"), _reply(path, "second answer"), _reply(path, "third answer")) + + with Redis(host=os.environ["REDIS_HOST"], port=int(os.environ["REDIS_PORT"])) as redis: + cached_before: Final = _cached_responses(redis) + gateway.post(path, _body(path, model, under_cap)) + eventually(lambda: _cached_responses(redis) - cached_before, lambda written: len(written) == 1) + repeated: Final = gateway.post(path, _body(path, model, under_cap)) + past_cap_twice: Final = ( + gateway.post(path, _body(path, model, past_cap)), + gateway.post(path, _body(path, model, past_cap)), + ) + + assert _answer(path, repeated) == "first answer", "a repeated turn under max_messages was not served from the cache" + assert tuple(_answer(path, answer) for answer in past_cap_twice) == ("second answer", "third answer"), ( + "a turn past max_messages was served from the cache" + ) + assert len(provider.received()) == 3 diff --git a/tests/unit/caching/test_caching.py b/tests/unit/caching/test_caching.py index 0a7ac3ecad1..0adb6b9a6f4 100644 --- a/tests/unit/caching/test_caching.py +++ b/tests/unit/caching/test_caching.py @@ -1,6 +1,7 @@ import asyncio import logging import re +import uuid from typing import Final from unittest.mock import MagicMock @@ -9,7 +10,7 @@ import pytest import litellm import litellm.caching.redis_cache as redis_cache_module from litellm._internal_context import current_service_target -from litellm.caching.caching import Cache, response_cache_phase +from litellm.caching.caching import Cache, CacheMode, response_cache_phase from litellm.caching.caching_handler import _PENDING_CACHE_WRITES from litellm.caching.in_memory_cache import InMemoryCache from litellm.caching.redis_cache import RedisCache, _RedisTimeoutLogThrottle @@ -473,3 +474,65 @@ async def test_a_lookup_already_inside_the_phase_does_not_open_a_second_one(v2_s await cache.async_get_cache(dynamic_cache_object=backend, **_REQUEST) assert backend.seen == [("llm_response", "cache.get llm_response")] assert [s.name for s in v2_span_exporter.get_finished_spans()] == ["cache.get llm_response"] + + +_TOOL_TURN_ITEM: Final = {"role": "user", "content": "hi"} + + +@pytest.mark.parametrize( + ("kwargs", "expected"), + [ + pytest.param({"messages": [_TOOL_TURN_ITEM] * 4}, True, id="four-messages-are-cached"), + pytest.param({"messages": [_TOOL_TURN_ITEM] * 5}, False, id="five-messages-skip-the-cache"), + pytest.param({"input": [_TOOL_TURN_ITEM] * 4}, True, id="four-responses-items-are-cached"), + pytest.param({"input": [_TOOL_TURN_ITEM] * 5}, False, id="five-responses-items-skip-the-cache"), + pytest.param({"input": "one prompt"}, True, id="string-input-is-one-message"), + pytest.param({"input": ["a", "b", "c", "d", "e"]}, True, id="embedding-strings-are-not-messages"), + ], +) +def test_should_use_cache_stops_past_the_default_max_messages(kwargs: dict[str, object], expected: bool) -> None: + assert Cache(type=LiteLLMCacheType.LOCAL).should_use_cache(**kwargs) is expected + + +def test_responses_sdk_items_count_toward_max_messages() -> None: + from openai.types.responses import ResponseFunctionToolCall + + call: Final = ResponseFunctionToolCall(type="function_call", call_id="c1", name="ls", arguments="{}") + + assert Cache(type=LiteLLMCacheType.LOCAL).should_use_cache(input=[_TOOL_TURN_ITEM, call, call, call, call]) is False + + +def test_max_messages_is_configurable_and_none_disables_it() -> None: + three: Final = [_TOOL_TURN_ITEM] * 3 + + assert Cache(type=LiteLLMCacheType.LOCAL, max_messages=2).should_use_cache(messages=three) is False + assert Cache(type=LiteLLMCacheType.LOCAL, max_messages=3).should_use_cache(messages=three) is True + assert Cache(type=LiteLLMCacheType.LOCAL, max_messages=None).should_use_cache(messages=three * 50) is True + + +def test_max_messages_beats_an_explicit_use_cache_opt_in() -> None: + cache: Final = Cache(type=LiteLLMCacheType.LOCAL, mode=CacheMode.default_off) + + assert cache.should_use_cache(messages=[_TOOL_TURN_ITEM] * 4, cache={"use-cache": True}) is True + assert cache.should_use_cache(messages=[_TOOL_TURN_ITEM] * 5, cache={"use-cache": True}) is False + + +def test_completion_past_max_messages_is_neither_served_from_nor_written_to_the_cache( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(litellm, "cache", Cache(type=LiteLLMCacheType.LOCAL)) + tag: Final = uuid.uuid4().hex + four: Final = [{"role": "user", "content": f"{tag} turn {index}"} for index in range(4)] + five: Final = [*four, {"role": "user", "content": f"{tag} turn 4"}] + + def answer(messages: list[dict[str, str]], mock_response: str) -> str: + response: Final = litellm.completion(model="gpt-4o-mini", messages=messages, mock_response=mock_response) + assert isinstance(response, litellm.ModelResponse), response + choice: Final = response.choices[0] + assert isinstance(choice, litellm.Choices), choice + return str(choice.message.content) + + assert answer(four, "four first") == "four first" + assert answer(four, "four second") == "four first", "a 4-message repeat missed the cache" + assert answer(five, "five first") == "five first" + assert answer(five, "five second") == "five second", "a 5-message repeat was served from the cache" diff --git a/tests/unit/caching/test_redis_semantic_cache.py b/tests/unit/caching/test_redis_semantic_cache.py index 99844c695cb..93db26fc9b2 100644 --- a/tests/unit/caching/test_redis_semantic_cache.py +++ b/tests/unit/caching/test_redis_semantic_cache.py @@ -568,6 +568,20 @@ def test_redis_semantic_cache_set_cache_flattens_structured_responses_input(): ) +def test_redis_semantic_cache_prompt_extraction_reads_function_call_output_blocks(): + from litellm.caching.redis_semantic_cache import RedisSemanticCache + + prompt = RedisSemanticCache._get_prompt_from_kwargs( + input=[ + {"role": "user", "content": "update the config"}, + {"type": "function_call", "call_id": "c1", "name": "write_file", "arguments": '{"path": "a"}'}, + {"type": "function_call_output", "call_id": "c1", "output": [{"type": "input_text", "text": "wrote a"}]}, + ] + ) + + assert prompt == "update the config\nwrote a" + + def test_redis_semantic_cache_prompt_extraction_prefers_messages(): from litellm.caching.redis_semantic_cache import RedisSemanticCache @@ -1416,3 +1430,23 @@ async def test_redis_async_embedding_truncates_off_the_event_loop(monkeypatch): assert embedding == [0.1, 0.2] assert _token_count("sem-embed", router.aembedding.call_args.kwargs["input"]) == 5 assert_loop_stayed_free(took, lags) + + +def test_redis_semantic_cache_prompt_extraction_keeps_tool_result_text(): + from litellm.caching.redis_semantic_cache import RedisSemanticCache + + prompt = RedisSemanticCache._get_prompt_from_kwargs( + messages=[ + {"role": "user", "content": "list the files"}, + { + "role": "assistant", + "content": [{"type": "tool_use", "id": "toolu_1", "name": "Bash", "input": {"command": "ls"}}], + }, + { + "role": "user", + "content": [{"type": "tool_result", "tool_use_id": "toolu_1", "content": "calc.py test_calc.py"}], + }, + ] + ) + + assert prompt == "list the filescalc.py test_calc.py" diff --git a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_common_utils.py b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_common_utils.py index 7415c74226d..1d19264fb9e 100644 --- a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_common_utils.py +++ b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_common_utils.py @@ -16,6 +16,8 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import ( encrypted_reasoning_signature, get_file_ids_from_messages, get_format_from_file_id, + get_semantic_cache_prompt_from_messages, + get_str_from_messages, handle_any_messages_to_chat_completion_str_messages_conversion, hoist_images_from_tool_messages, is_encrypted_reasoning_block, @@ -2171,3 +2173,93 @@ class TestMergeConsecutiveSystemMessages: ) assert merged == [{"role": "system"}, {"role": "user", "content": "Hi"}] + + +_CLAUDE_CODE_TOOL_TURN: Final = [ + {"role": "user", "content": "list the files"}, + { + "role": "assistant", + "content": [{"type": "tool_use", "id": "toolu_1", "name": "Bash", "input": {"command": "ls"}}], + }, + { + "role": "user", + "content": [{"type": "tool_result", "tool_use_id": "toolu_1", "content": "calc.py test_calc.py"}], + }, +] + + +@pytest.mark.parametrize( + ("messages", "expected"), + [ + pytest.param(_CLAUDE_CODE_TOOL_TURN, "list the filescalc.py test_calc.py", id="tool-result-string"), + pytest.param( + [ + { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "toolu_1", + "content": [ + {"type": "text", "text": "x = 1"}, + {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": ""}}, + ], + } + ], + } + ], + "x = 1", + id="tool-result-blocks", + ), + pytest.param( + [{"role": "user", "content": [{"type": "tool_result", "tool_use_id": "toolu_1"}]}], + "", + id="tool-result-without-content", + ), + ], +) +def test_get_semantic_cache_prompt_from_messages_keeps_tool_result_text( + messages: list[dict[str, object]], expected: str +) -> None: + assert get_semantic_cache_prompt_from_messages(messages) == expected + + +def test_get_semantic_cache_prompt_from_messages_differs_from_the_turn_before_it() -> None: + assert get_str_from_messages(_CLAUDE_CODE_TOOL_TURN) == get_str_from_messages(_CLAUDE_CODE_TOOL_TURN[:1]) + assert get_semantic_cache_prompt_from_messages(_CLAUDE_CODE_TOOL_TURN) != get_semantic_cache_prompt_from_messages( + _CLAUDE_CODE_TOOL_TURN[:1] + ) + + +@pytest.mark.parametrize( + "messages", + [ + pytest.param([{"role": "system", "content": "be brief. "}, {"role": "user", "content": "hello"}], id="strings"), + pytest.param( + [ + { + "role": "user", + "content": [ + {"type": "text", "text": "What is "}, + {"type": "image_url", "image_url": {"url": "https://example.com/a.png"}}, + {"type": "text", "text": "this?"}, + ], + } + ], + id="text-parts", + ), + pytest.param( + [ + {"role": "assistant"}, + {"role": "assistant", "content": None}, + {"role": "user", "content": ""}, + {"role": "tool", "content": "small", "search_results": [{"source": "s", "title": "t", "content": []}]}, + ], + id="empty-content-and-search-results", + ), + ], +) +def test_get_semantic_cache_prompt_from_messages_matches_get_str_from_messages_without_tool_results( + messages: list[dict[str, object]], +) -> None: + assert get_semantic_cache_prompt_from_messages(messages) == get_str_from_messages(messages) From c943d650f40a4c32f25d52706f6ef88dd6ee9805 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Wed, 7 Oct 2026 10:32:58 -0700 Subject: [PATCH 07/13] test(e2e): tag router, batches and mcp tests with Subject metadata and record client steps (#44964) * test(e2e): add enum values, auto-discovering label gates and secret hiding for e2e metadata * test(e2e): tag router, batches and mcp tests with Subject metadata and record client steps * test(e2e): leave the batches cleanup harness unit tests untagged * docs(e2e): name every markerless harness test file that carries no Subject * test(e2e): keep the step discovery comprehensions to one for clause * test(e2e): name every driven model on the vllm batch, prompt caching and complexity router subjects --- tests/e2e/batches/batch_cleanup.py | 7 +- tests/e2e/batches/batch_client.py | 12 + tests/e2e/batches/capabilities.py | 42 +++- tests/e2e/batches/test_batches_e2e.py | 206 +++++++++++++++++- .../test_managed_files_enforcement_e2e.py | 29 ++- tests/e2e/mcp/conftest.py | 5 +- tests/e2e/mcp/datadog_mcp.py | 7 + tests/e2e/mcp/mcp_client.py | 16 ++ tests/e2e/mcp/oauth_chat_client.py | 9 + tests/e2e/mcp/oauth_gateway.py | 6 + tests/e2e/mcp/test_mcp_access_group_e2e.py | 7 + .../mcp/test_mcp_chat_completion_oauth_e2e.py | 21 ++ tests/e2e/mcp/test_mcp_datadog_e2e.py | 13 +- tests/e2e/mcp/test_mcp_guardrail_e2e.py | 7 + tests/e2e/mcp/test_mcp_key_access_e2e.py | 25 +++ .../e2e/mcp/test_mcp_oauth_happy_path_e2e.py | 7 + .../mcp/test_mcp_toolset_enforcement_e2e.py | 7 + tests/e2e/router/reliability_support.py | 28 +++ .../test_auto_router_regressions_e2e.py | 112 ++++++++++ .../e2e/router/test_complexity_router_e2e.py | 17 +- .../e2e/router/test_reliability_cache_e2e.py | 13 +- ...st_reliability_cancel_on_disconnect_e2e.py | 10 + .../router/test_reliability_cooldowns_e2e.py | 42 ++++ .../router/test_reliability_fallbacks_e2e.py | 46 +++- .../e2e/router/test_reliability_memory_e2e.py | 16 +- .../test_reliability_prompt_caching_e2e.py | 11 + .../router/test_reliability_retries_e2e.py | 43 ++++ ...test_reliability_routing_strategies_e2e.py | 40 ++++ .../router/test_reliability_timeouts_e2e.py | 5 +- 29 files changed, 773 insertions(+), 36 deletions(-) diff --git a/tests/e2e/batches/batch_cleanup.py b/tests/e2e/batches/batch_cleanup.py index f1142a60782..f8fc1bbf01f 100644 --- a/tests/e2e/batches/batch_cleanup.py +++ b/tests/e2e/batches/batch_cleanup.py @@ -8,6 +8,7 @@ from typing import Final, Protocol from batch_client import BatchObject, FileDeleteResponse from capabilities import is_cloud_storage_id, is_managed_id from e2e_http import NetworkError, RateLimitedError, Result, Success, UnknownApiError +from e2e_metadata import STEP_FRAMES, step from pydantic import BaseModel CLEANUP_DELAYS: Final = (1.0, 2.0, 4.0) @@ -52,6 +53,7 @@ def _require_cleanup_success[R: BaseModel](result: Result[R], operation: str) -> raise AssertionError(f"{operation} failed: {result.kind}") +@step("Clean up the uploaded file") def cleanup_file(client: BatchCleanupClient, file_id: str, *, key: str, provider: str | None = None) -> None: delete: Final[Callable[[], Result[FileDeleteResponse]]] = ( (lambda: client.delete_file_as_admin(file_id, provider=provider)) @@ -65,7 +67,7 @@ def cleanup_file(client: BatchCleanupClient, file_id: str, *, key: str, provider warnings.warn( f"Left file {file_id} in place: LiteLLM refused to delete it while a batch still references it", UserWarning, - stacklevel=2, + stacklevel=2 + STEP_FRAMES, ) return deleted: Final = _require_cleanup_success(result, f"Delete file {file_id}") @@ -74,6 +76,7 @@ def cleanup_file(client: BatchCleanupClient, file_id: str, *, key: str, provider ), f"Delete file {file_id} did not confirm deletion" +@step("Cancel the batch if it is still running") def cleanup_batch( client: BatchCleanupClient, batch_id: str, @@ -137,7 +140,7 @@ def cleanup_batch( warnings.warn( f"Left batch {batch_id} cancelling after {BATCH_CANCEL_TIMEOUT_SECONDS}s for the provider to finish", UserWarning, - stacklevel=2, + stacklevel=2 + STEP_FRAMES, ) return wait(BATCH_CANCEL_POLL_SECONDS) diff --git a/tests/e2e/batches/batch_client.py b/tests/e2e/batches/batch_client.py index 8745140a818..b02b09e557b 100644 --- a/tests/e2e/batches/batch_client.py +++ b/tests/e2e/batches/batch_client.py @@ -17,6 +17,7 @@ from typing import Final, Literal from pydantic import BaseModel, Field +from e2e_metadata import step from proxy_client import ProxyClient from e2e_http import ( FileUploadForm, @@ -136,12 +137,15 @@ def is_result_access_denied[R: BaseModel](result: Result[R]) -> bool: class BatchClient: proxy: ProxyClient + @step("Add a batch deployment named {model_name} that calls {litellm_params.model}") def create_model(self, model_name: str, litellm_params: LiteLLMParamsBody) -> str: return self.proxy.create_model(model_name, litellm_params, mode="batch") + @step("Delete the batch deployment") def delete_model(self, model_id: str) -> None: self.proxy.delete_model(model_id) + @step("Upload a batch input file to /v1/files") def upload_file( self, *, @@ -161,6 +165,7 @@ class BatchClient: response_type=FileObject, ) + @step("Retrieve the uploaded file") def retrieve_file( self, file_id: str, *, key: str, provider: str | None = None ) -> Result[FileObject]: @@ -171,6 +176,7 @@ class BatchClient: response_type=FileObject, ) + @step("List the files the key can see from /v1/files") def list_files(self, *, key: str, provider: str | None = None) -> Result[FileList]: return self.proxy.transport.get( _files_path(provider), @@ -179,6 +185,7 @@ class BatchClient: response_type=FileList, ) + @step("Create a batch of {body.endpoint} requests from the uploaded file") def create_batch( self, *, body: BatchCreateBody, key: str, provider: str | None = None ) -> StreamingResponse: @@ -188,6 +195,7 @@ class BatchClient: json=body, ) + @step("Retrieve the batch") def retrieve_batch( self, batch_id: str, *, key: str, provider: str | None = None ) -> Result[BatchObject]: @@ -198,6 +206,7 @@ class BatchClient: response_type=BatchObject, ) + @step("Cancel the batch") def cancel_batch( self, batch_id: str, *, key: str, provider: str | None = None ) -> Result[BatchObject]: @@ -208,6 +217,7 @@ class BatchClient: response_type=BatchObject, ) + @step("List the batches the key can see from /v1/batches") def list_batches( self, *, @@ -223,6 +233,7 @@ class BatchClient: response_type=BatchList, ) + @step("Delete the uploaded file") def delete_file( self, file_id: str, *, key: str, provider: str | None = None ) -> Result[FileDeleteResponse]: @@ -233,6 +244,7 @@ class BatchClient: response_type=FileDeleteResponse, ) + @step("Delete the uploaded file as the proxy admin") def delete_file_as_admin(self, file_id: str, *, provider: str | None = None) -> Result[FileDeleteResponse]: return self.proxy.transport.delete( f"{_files_path(provider)}/{file_id}", diff --git a/tests/e2e/batches/capabilities.py b/tests/e2e/batches/capabilities.py index d510426dee2..c03b481f060 100644 --- a/tests/e2e/batches/capabilities.py +++ b/tests/e2e/batches/capabilities.py @@ -7,7 +7,11 @@ import os from dataclasses import dataclass from typing import Final, Literal +import pytest + from e2e_config import provider_edge_base, unique_marker +from e2e_metadata import Domain, Mode, Route, Subject, meta +from e2e_metadata import Provider as MetaProvider from models import LiteLLMParamsBody _BATCH_RUN = unique_marker() @@ -18,6 +22,9 @@ def batch_model_name(base: str) -> str: OPENAI_BATCH_BACKEND: Final = "gpt-4o-mini" +AZURE_BATCH_BACKEND: Final = "gpt-5.4-mini-batch" +VERTEX_BATCH_BACKEND: Final = "gemini-2.5-flash" +BEDROCK_BATCH_BACKEND: Final = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0" def openai_batch_params() -> LiteLLMParamsBody: @@ -65,14 +72,14 @@ class Provider: return openai_batch_params() case "azure": return LiteLLMParamsBody( - model="azure/gpt-5.4-mini-batch", + model=f"azure/{AZURE_BATCH_BACKEND}", api_base="os.environ/AZURE_API_BASE", api_key="os.environ/AZURE_API_KEY", api_version="2025-04-01-preview", ) case "vertex_ai": return LiteLLMParamsBody( - model="vertex_ai/gemini-2.5-flash", + model=f"vertex_ai/{VERTEX_BATCH_BACKEND}", vertex_project="os.environ/VERTEXAI_PROJECT", vertex_location="us-central1", vertex_credentials="os.environ/VERTEXAI_CREDENTIALS", @@ -81,7 +88,7 @@ class Provider: ) case "bedrock": return LiteLLMParamsBody( - model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", + model=BEDROCK_BATCH_BACKEND, aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID", aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY", aws_region_name="os.environ/AWS_REGION", @@ -132,21 +139,21 @@ PROVIDERS: tuple[Provider, ...] = ( Provider( "azure", batch_model_name("azure-batch"), - "gpt-5.4-mini-batch", + AZURE_BATCH_BACKEND, can_cancel=True, can_list=True, ), Provider( "vertex_ai", batch_model_name("vertex-batch"), - "gemini-2.5-flash", + VERTEX_BATCH_BACKEND, can_cancel=True, can_list=True, ), Provider( "bedrock", batch_model_name("bedrock-batch"), - "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", + BEDROCK_BATCH_BACKEND, can_cancel=True, can_list=True, ), @@ -181,6 +188,29 @@ CAPABILITIES: tuple[Capability, ...] = tuple( ) +def lifecycle_meta(cap: Capability) -> pytest.MarkDecorator: + return meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider(cap.provider),), + models=(cap.raw_model,), + mode=Mode.BATCH, + ) + ) + + +def file_content_meta(provider: Provider) -> pytest.MarkDecorator: + return meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + providers=(MetaProvider(provider.name),), + models=(provider.raw_model,), + ) + ) + + def raw_id_matches_provider(provider: str, batch_id: str) -> bool: if provider in ("openai", "azure"): return batch_id.startswith("batch") diff --git a/tests/e2e/batches/test_batches_e2e.py b/tests/e2e/batches/test_batches_e2e.py index 8da2deb4010..54bcb80ed24 100644 --- a/tests/e2e/batches/test_batches_e2e.py +++ b/tests/e2e/batches/test_batches_e2e.py @@ -44,17 +44,22 @@ from capabilities import ( OPENAI_BATCH_BACKEND, OPENAI_BATCH_MODEL, PROVIDERS, + VERTEX_BATCH_BACKEND, Capability, Provider, batch_model_name, coverage_cells_for_lifecycle, decoded_model_from_id, + file_content_meta, is_managed_id, + lifecycle_meta, matches_id_shape, openai_batch_params, raw_id_matches_provider, ) from e2e_config import MASTER_KEY, PROXY_BASE_URL, unique_marker +from e2e_metadata import Domain, Mode, Route, Subject, meta +from e2e_metadata import Provider as MetaProvider from e2e_http import ( FileUploadForm, Result, @@ -249,7 +254,7 @@ def assert_batch_object(batch: BatchObject) -> None: pytest.param( cap, id=cap.id, - marks=pytest.mark.covers(*coverage_cells_for_lifecycle(cap)), + marks=(pytest.mark.covers(*coverage_cells_for_lifecycle(cap)), lifecycle_meta(cap)), ) for cap in CAPABILITIES ], @@ -350,6 +355,15 @@ def test_batch_lifecycle( @pytest.mark.covers("llm.batches.openai.key_model_access_denied.nonstream.works") +@meta( + Subject( + domain=Domain.PROXY_AUTH, + route=Route.BATCHES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + mode=Mode.BATCH, + ) +) def test_batch_key_model_access_denied( client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -389,6 +403,14 @@ def test_batch_key_model_access_denied( "llm.files.openai.upload.nonstream.works", "llm.files.openai.delete.nonstream.works", ) +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + ) +) def test_file_upload_and_delete_outputs( client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -433,6 +455,15 @@ def unattributed_rows(rows: list[SpendLogRow]) -> list[SpendLogRow]: "once the fetch is bounded." ) ) +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.BATCHES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + mode=Mode.BATCH, + ) +) def test_rate_limited_batch_create_leaves_no_unattributed_spend_row( client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -471,7 +502,7 @@ def test_rate_limited_batch_create_leaves_no_unattributed_spend_row( file = unwrap( client.upload_file( - content=render_jsonl("gpt-4o-mini"), + content=render_jsonl(OPENAI_BATCH_BACKEND), form=FileUploadForm(purpose="batch"), model=OPENAI_BATCH_MODEL, key=key, @@ -520,6 +551,14 @@ class TestBatchFileContent: "llm.files.openai.content.nonstream.works", exercised_on=["files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + ) + ) def test_file_content_matches_upload( self, client: BatchClient, resources: ResourceManager ) -> None: @@ -558,8 +597,9 @@ class TestBatchFileContent: pytest.param( p, id=p.name, - marks=pytest.mark.covers( - FILE_CONTENT_CELLS[p.name], exercised_on=["files"] + marks=( + pytest.mark.covers(FILE_CONTENT_CELLS[p.name], exercised_on=["files"]), + file_content_meta(p), ), ) for p in PROVIDERS @@ -632,6 +672,14 @@ class TestOpenAIFiles: "marker when LIT-4820 is fixed; do not relax the assertion to make it pass." ) ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + ) + ) def test_uploaded_file_appears_in_list( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -662,6 +710,7 @@ class TestOpenAIFiles: "llm.files.openai.list_isolation.nonstream.works", exercised_on=["files"], ) + @meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.FILES, providers=(MetaProvider.OPENAI,))) def test_list_page_cursors_address_only_the_callers_own_files( self, client: BatchClient, resources: ResourceManager ) -> None: @@ -697,6 +746,14 @@ class TestOpenAIFiles: "llm.files.openai.retrieve.nonstream.works", exercised_on=["files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + ) + ) def test_retrieve_round_trips_metadata( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -760,6 +817,15 @@ class TestBatchRateLimitErrorMapping: "quota_management.ratelimit.batch_rpm.blocks_over_limit", exercised_on=["batches"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.BATCHES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + mode=Mode.BATCH, + ) + ) def test_batch_create_over_rpm_returns_mapped_429( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -773,7 +839,7 @@ class TestBatchRateLimitErrorMapping: file = unwrap( client.upload_file( - content=_multi_request_jsonl("gpt-4o-mini", BATCH_RL_REQUEST_LINES), + content=_multi_request_jsonl(OPENAI_BATCH_BACKEND, BATCH_RL_REQUEST_LINES), form=FileUploadForm(purpose="batch"), model=OPENAI_BATCH_MODEL, key=key, @@ -826,7 +892,7 @@ class TestBatchEnqueuedTokenLimit: ) -> FileObject: file = unwrap( client.upload_file( - content=_multi_request_jsonl("gpt-4o-mini", BATCH_RL_REQUEST_LINES), + content=_multi_request_jsonl(OPENAI_BATCH_BACKEND, BATCH_RL_REQUEST_LINES), form=FileUploadForm(purpose="batch"), model=OPENAI_BATCH_MODEL, key=key, @@ -859,6 +925,15 @@ class TestBatchEnqueuedTokenLimit: "quota_management.ratelimit.batch_enqueued_tokens.accepts_over_rpm", exercised_on=["batches"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.BATCHES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + mode=Mode.BATCH, + ) + ) def test_enqueued_allowance_accepts_batch_over_key_rpm( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -890,6 +965,15 @@ class TestBatchEnqueuedTokenLimit: "quota_management.ratelimit.batch_enqueued_tokens.refunds_on_cancel", exercised_on=["batches"], ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.BATCHES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + mode=Mode.BATCH, + ) + ) def test_exhausted_allowance_blocks_until_cancel_refunds( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -985,6 +1069,15 @@ class TestBedrockBatchAssumeRole: "llm.files.bedrock.upload.nonstream.works", exercised_on=["batches", "files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.BEDROCK,), + models=(ASSUME_ROLE_RAW_MODEL,), + mode=Mode.BATCH, + ) + ) def test_unified_batch_create_with_assume_role( self, client: BatchClient, resources: ResourceManager ) -> None: @@ -1052,6 +1145,14 @@ class TestBedrockBatchSplitS3Credentials: "llm.files.bedrock.split_s3_credentials.nonstream.works", exercised_on=["files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + providers=(MetaProvider.BEDROCK,), + models=(ASSUME_ROLE_RAW_MODEL,), + ) + ) def test_file_lifecycle_signs_s3_with_s3_credentials( self, client: BatchClient, resources: ResourceManager ) -> None: @@ -1123,6 +1224,15 @@ class TestBedrockBatchGovCloud: "llm.files.bedrock.govcloud_partition.nonstream.works", exercised_on=["batches", "files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.BEDROCK,), + models=(GOVCLOUD_RAW_MODEL,), + mode=Mode.BATCH, + ) + ) def test_unified_file_upload_and_batch_create_in_govcloud( self, client: BatchClient, resources: ResourceManager ) -> None: @@ -1193,6 +1303,14 @@ class TestGeminiFiles: "llm.files.gemini.upload.nonstream.works", exercised_on=["files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + providers=(MetaProvider.GEMINI,), + models=(GEMINI_FILES_RAW_MODEL,), + ) + ) def test_gemini_file_upload( self, client: BatchClient, resources: ResourceManager ) -> None: @@ -1227,7 +1345,7 @@ def _vllm_params(api_base: str, api_key: str | None, model_id: str) -> LiteLLMPa ) -HOSTED_VLLM_DEFAULT_MODEL = "Qwen/Qwen2.5-0.5B-Instruct" +HOSTED_VLLM_MODEL: Final = (os.environ.get("HOSTED_VLLM_MODEL") or "Qwen/Qwen2.5-0.5B-Instruct").strip() HOSTED_VLLM_BAD_LINE_CUSTOM_ID = "req-bad" @@ -1236,9 +1354,8 @@ def _hosted_vllm_deployment(client: BatchClient, resources: ResourceManager) -> if api_base is None: pytest.skip("set HOSTED_VLLM_API_BASE (the live vLLM server this deployment targets)") api_key = (os.environ.get("HOSTED_VLLM_API_KEY") or "").strip() or None - model_id = (os.environ.get("HOSTED_VLLM_MODEL") or HOSTED_VLLM_DEFAULT_MODEL).strip() proxy_name = batch_model_name("hosted-vllm-batch") - model_row_id = client.create_model(proxy_name, _vllm_params(api_base, api_key, model_id)) + model_row_id = client.create_model(proxy_name, _vllm_params(api_base, api_key, HOSTED_VLLM_MODEL)) resources.defer(lambda: client.delete_model(model_row_id)) return proxy_name @@ -1290,6 +1407,15 @@ class TestHostedVllmBatch: "llm.files.hosted_vllm.upload.nonstream.works", exercised_on=["batches", "files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.HOSTED_VLLM,), + models=(HOSTED_VLLM_MODEL,), + mode=Mode.BATCH, + ) + ) def test_batch_runs_to_completion_with_a_downloadable_output( self, client: BatchClient, resources: ResourceManager, upload_route: str ) -> None: @@ -1337,6 +1463,15 @@ class TestHostedVllmBatch: ) @pytest.mark.covers("llm.batches.hosted_vllm.basic.nonstream.works", exercised_on=["batches", "files"]) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.HOSTED_VLLM,), + models=(HOSTED_VLLM_MODEL,), + mode=Mode.BATCH, + ) + ) def test_failing_line_lands_in_the_error_file_not_the_batch_status( self, client: BatchClient, resources: ResourceManager ) -> None: @@ -1417,6 +1552,7 @@ class TestBatchFailurePaths: "llm.batches.openai.malformed_jsonl.nonstream.works", exercised_on=["files"], ) + @meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.FILES)) def test_malformed_jsonl_upload_rejected( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -1439,13 +1575,22 @@ class TestBatchFailurePaths: "llm.batches.openai.cancel_terminal.nonstream.works", exercised_on=["batches", "files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + mode=Mode.BATCH, + ) + ) def test_endpoint_mismatch_fails_batch_and_cancel_conflicts( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: key = resources.key() file = unwrap( client.upload_file( - content=_mismatched_endpoint_jsonl("gpt-4o-mini"), + content=_mismatched_endpoint_jsonl(OPENAI_BATCH_BACKEND), form=FileUploadForm(purpose="batch"), model=OPENAI_BATCH_MODEL, key=key, @@ -1495,6 +1640,15 @@ class TestBatchFailurePaths: "llm.batches.openai.foreign_file_id.nonstream.works", exercised_on=["batches", "files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.AZURE,), + models=(AZURE_BATCH_RAW_MODEL,), + mode=Mode.BATCH, + ) + ) def test_foreign_encoded_file_id_routes_by_file_model( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -1544,6 +1698,15 @@ class TestBatchSecondHop: "llm.batches.openai.second_hop.nonstream.works", exercised_on=["batches", "files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.LITELLM_PROXY, MetaProvider.OPENAI), + models=(OPENAI_BATCH_BACKEND,), + mode=Mode.BATCH, + ) + ) def test_unified_create_and_retrieve_via_chained_gateway( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -1561,7 +1724,7 @@ class TestBatchSecondHop: file = unwrap( client.upload_file( - content=render_jsonl("gpt-4o-mini"), + content=render_jsonl(OPENAI_BATCH_BACKEND), form=FileUploadForm(purpose="batch", target_model_names=hop_name), key=key, ) @@ -1680,13 +1843,22 @@ class TestBatchTerminalState: "llm.batches.openai.terminal_state.nonstream.cost_logged", exercised_on=["batches", "files"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + mode=Mode.BATCH, + ) + ) def test_completed_batch_downloads_output_and_books_cost( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: key = resources.key() file = unwrap( client.upload_file( - content=render_jsonl("gpt-4o-mini"), + content=render_jsonl(OPENAI_BATCH_BACKEND), form=FileUploadForm(purpose="batch"), model=OPENAI_BATCH_MODEL, key=key, @@ -1786,6 +1958,15 @@ class TestVertexNativePassthrough: "llm.batches.vertex.native_passthrough.nonstream.works", exercised_on=["files", "batches"], ) + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.BATCHES, + providers=(MetaProvider.VERTEX_AI,), + models=(VERTEX_BATCH_BACKEND,), + mode=Mode.BATCH, + ) + ) def test_native_jsonl_round_trips_untouched_and_starts_a_batch( self, client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: @@ -1848,6 +2029,7 @@ class TestVertexNativePassthrough: ), ], ) + @meta(Subject(domain=Domain.LLM_TRANSLATION, route=Route.FILES)) def test_passthrough_upload_is_rejected_outside_a_native_vertex_batch( self, content: bytes, diff --git a/tests/e2e/batches/test_managed_files_enforcement_e2e.py b/tests/e2e/batches/test_managed_files_enforcement_e2e.py index 2f5d0588aca..43a2488289c 100644 --- a/tests/e2e/batches/test_managed_files_enforcement_e2e.py +++ b/tests/e2e/batches/test_managed_files_enforcement_e2e.py @@ -22,9 +22,10 @@ import pytest from batch_client import BatchClient, FileObject from batch_cleanup import cleanup_file -from capabilities import batch_model_name, is_managed_id, openai_batch_params +from capabilities import OPENAI_BATCH_BACKEND, batch_model_name, is_managed_id, openai_batch_params from e2e_config import unique_marker from e2e_http import FileUploadForm, Result, UnknownApiError, unwrap +from e2e_metadata import Domain, Provider, Route, Subject, meta from lifecycle import ResourceManager pytestmark = [pytest.mark.e2e, pytest.mark.managed_files] @@ -64,6 +65,12 @@ def managed_model(client: BatchClient) -> Iterator[str]: @pytest.mark.covers(UPLOAD_ROW) +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + ) +) def test_upload_without_target_model_names_rejected( client: BatchClient, scoped_key: str, managed_model: str ) -> None: @@ -76,6 +83,12 @@ def test_upload_without_target_model_names_rejected( @pytest.mark.covers(UPLOAD_ROW) +@meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.FILES, + ) +) def test_upload_with_model_param_rejected( client: BatchClient, scoped_key: str, managed_model: str ) -> None: @@ -89,12 +102,26 @@ def test_upload_with_model_param_rejected( @pytest.mark.covers(ISOLATION_ROW) +@meta( + Subject( + domain=Domain.PROXY_AUTH, + route=Route.FILES, + ) +) def test_raw_provider_file_id_rejected(client: BatchClient, scoped_key: str) -> None: result = client.retrieve_file("file-e2e-raw-provider-id", key=scoped_key) expect_api_error(result, 400, "Raw provider file ids cannot be used") @pytest.mark.covers(ISOLATION_ROW) +@meta( + Subject( + domain=Domain.PROXY_AUTH, + route=Route.FILES, + providers=(Provider.OPENAI,), + models=(OPENAI_BATCH_BACKEND,), + ) +) def test_cross_user_managed_id_denied_owner_allowed( client: BatchClient, resources: ResourceManager, managed_model: str ) -> None: diff --git a/tests/e2e/mcp/conftest.py b/tests/e2e/mcp/conftest.py index e6094ab95ea..91f55a528b5 100644 --- a/tests/e2e/mcp/conftest.py +++ b/tests/e2e/mcp/conftest.py @@ -15,14 +15,11 @@ from typing import Protocol, cast import pytest +from datadog_mcp import DdLogsReader from mcp_client import McpClient, build_client from proxy_client import ProxyClient -class DdLogsReader(Protocol): - def poll_events_for_marker(self, marker: str) -> list[object]: ... - - class _DdLogsReaderBuilder(Protocol): def __call__(self) -> DdLogsReader: ... diff --git a/tests/e2e/mcp/datadog_mcp.py b/tests/e2e/mcp/datadog_mcp.py index 352b4446cfd..63194af221e 100644 --- a/tests/e2e/mcp/datadog_mcp.py +++ b/tests/e2e/mcp/datadog_mcp.py @@ -4,14 +4,20 @@ from __future__ import annotations import os from collections.abc import Sequence +from typing import Protocol from e2e_config import datadog_mcp_url, unique_marker +from e2e_metadata import step from lifecycle import ResourceManager from mcp_client import McpClient SEARCH_LOGS_TOOL = "search_datadog_logs" +class DdLogsReader(Protocol): + def poll_events_for_marker(self, marker: str) -> list[object]: ... + + def _dd_api_key() -> str: return os.environ.get("DD_API_KEY", "").strip() @@ -31,6 +37,7 @@ def assert_dd_mcp_creds() -> None: ) +@step("Register the Datadog remote MCP server with its credentials from the environment") def register_datadog_mcp( client: McpClient, resources: ResourceManager, diff --git a/tests/e2e/mcp/mcp_client.py b/tests/e2e/mcp/mcp_client.py index 56f7fffba29..e373a8e31ae 100644 --- a/tests/e2e/mcp/mcp_client.py +++ b/tests/e2e/mcp/mcp_client.py @@ -21,6 +21,7 @@ from pydantic import BaseModel, ConfigDict, Field, RootModel from e2e_config import settle_propagation from e2e_http import Headers, NoBody, Result, Success, UnknownApiError, unwrap +from e2e_metadata import step from models import KeyGenerateBody, McpServerListResponse, McpServerRow, ObjectPermission from proxy_client import ProxyClient @@ -153,6 +154,7 @@ class McpCallToolResponse(BaseModel): class McpClient: proxy: ProxyClient + @step("Register the MCP server {server_name} with the alias {alias}") def register_server( self, *, @@ -183,6 +185,7 @@ class McpClient: ) ).server_id + @step("Delete the MCP server") def delete_server(self, server_id: str) -> None: _ = self.proxy.transport.delete( f"/v1/mcp/server/{server_id}", @@ -191,6 +194,7 @@ class McpClient: response_type=NoBody, ) + @step("List the MCP servers from /v1/mcp/server") def registered_servers(self) -> list[McpServerRow]: return unwrap( self.proxy.transport.get( @@ -201,6 +205,7 @@ class McpClient: ) ).root + @step("List the MCP servers the key can see from /v1/mcp/server") def list_servers(self, key: str) -> Result[McpServerListResponse]: return self.proxy.transport.get( "/v1/mcp/server", @@ -209,6 +214,7 @@ class McpClient: response_type=McpServerListResponse, ) + @step("Check the health of the MCP servers the key can see from /v1/mcp/server/health") def server_health(self, key: str, server_ids: list[str] | None = None) -> Result[McpHealthResponse]: return self.proxy.transport.get( "/v1/mcp/server/health", @@ -217,6 +223,7 @@ class McpClient: response_type=McpHealthResponse, ) + @step("Wait for every proxy replica to list the MCP server in /v1/mcp/server") def await_registered(self, server_id: str) -> McpServerRow: """Wait for every configured replica to list the server and return its row.""" registered = self.proxy.read_body_back_everywhere( @@ -228,6 +235,7 @@ class McpClient: row for response in registered.values() for row in response.root if row.server_id == server_id ) + @step("Generate a virtual key for the user {user_id}") def generate_key( self, *, @@ -254,6 +262,7 @@ class McpClient: ) ) + @step("List the MCP tools the key can see from /mcp-rest/tools/list") def list_tools(self, key: str) -> Result[McpToolsListResponse]: return self.proxy.transport.get( "/mcp-rest/tools/list", @@ -262,6 +271,7 @@ class McpClient: response_type=McpToolsListResponse, ) + @step('Wait for /mcp-rest/tools/list to show the MCP server\'s tool matching "{needle}"') def await_tool(self, key: str, server_id: str, needle: str) -> str: """Poll tools/list until `server_id` serves a tool matching `needle`, and return its fully-qualified name. Fails at poll_timeout. @@ -287,6 +297,7 @@ class McpClient: ) time.sleep(self.proxy.poll_interval) + @step("Wait for /mcp-rest/tools/list to show the key exactly the expected tools on the MCP server") def await_tools(self, key: str, server_id: str, *, expected: frozenset[str]) -> frozenset[str]: """Poll tools/list until `server_id`'s tools as `key` sees them are exactly `expected`, and return the last listing either way, so the caller's equality @@ -301,6 +312,7 @@ class McpClient: return unwrap(result).tool_names_for_server(server_id) time.sleep(self.proxy.poll_interval) + @step("Call the MCP tool {name} through /mcp-rest/tools/call") def await_call_tool( self, key: str, @@ -329,6 +341,7 @@ class McpClient: ) time.sleep(self.proxy.poll_interval) + @step("Call the MCP tool {name} through /mcp-rest/tools/call and wait for a 403") def await_call_tool_denied( self, key: str, @@ -355,6 +368,7 @@ class McpClient: ) time.sleep(self.proxy.poll_interval) + @step('Create the guardrail {name} that blocks MCP tool calls containing "{blocked_keyword}"') def register_mcp_content_filter(self, *, name: str, blocked_keyword: str) -> str: """Register a default-on content-filter guardrail that runs on the MCP tool-call hook (pre_mcp_call) and blocks a single keyword. The keyword is @@ -378,6 +392,7 @@ class McpClient: settle_propagation(time.monotonic()) return guardrail_id + @step("Delete the guardrail") def delete_guardrail(self, guardrail_id: str) -> None: _ = self.proxy.transport.delete( f"/guardrails/{guardrail_id}", @@ -386,6 +401,7 @@ class McpClient: response_type=NoBody, ) + @step("Call the MCP tool {name} through /mcp-rest/tools/call with {arguments}") def call_tool( self, key: str, diff --git a/tests/e2e/mcp/oauth_chat_client.py b/tests/e2e/mcp/oauth_chat_client.py index 0c5c6106259..68e838d8745 100644 --- a/tests/e2e/mcp/oauth_chat_client.py +++ b/tests/e2e/mcp/oauth_chat_client.py @@ -26,6 +26,7 @@ import httpx2 import pytest from e2e_config import PROXY_BASE_URL, REQUEST_TIMEOUT from e2e_http import AuthHeaders, NoBody, unwrap +from e2e_metadata import step from idp import Identity from mcp import ClientSession from mcp.client.auth import OAuthClientProvider @@ -318,6 +319,7 @@ async def _list_and_call( class ChatMcpClient: proxy: ProxyClient + @step("Register the MCP server with the alias {body.alias}") def create_server(self, body: McpServerCreateBody) -> McpServerInfo: return unwrap( self.proxy.transport.post( @@ -328,6 +330,7 @@ class ChatMcpClient: ) ) + @step("Read the MCP server back from /v1/mcp/server") def server_info(self, server_id: str) -> McpServerInfo: return unwrap( self.proxy.transport.get( @@ -338,6 +341,7 @@ class ChatMcpClient: ) ) + @step("Delete the MCP server") def delete_server(self, server_id: str) -> None: _ = self.proxy.transport.delete( f"/v1/mcp/server/{server_id}", @@ -346,6 +350,7 @@ class ChatMcpClient: response_type=NoBody, ) + @step("Sign the key's user in to the MCP server {alias} through the OAuth consent flow") def seed_user_token(self, alias: str, key: str, storage_state_path: str) -> tuple[str, ...]: """Drive the interactive authorize dance for `key`'s user so the gateway stores their upstream token, retried to the shared deadline since the @@ -367,6 +372,7 @@ class ChatMcpClient: f"last error: {last_error!r}" ) + @step("List the tools on the MCP server {alias} and call {tool} over the MCP protocol") def list_and_call( self, alias: str, @@ -394,6 +400,7 @@ class ChatMcpClient: ) ) + @step("List the users with a stored OAuth token for the MCP server") def server_user_credentials(self, server_id: str) -> tuple[McpServerUserCredentialRow, ...]: return unwrap( self.proxy.transport.get( @@ -404,6 +411,7 @@ class ChatMcpClient: ) ).root + @step("Revoke the user's stored OAuth token for the MCP server") def revoke_user_token(self, server_id: str, headers: AuthHeaders) -> None: _ = unwrap( self.proxy.transport.delete( @@ -414,6 +422,7 @@ class ChatMcpClient: ) ) + @step("Send a /chat/completions request to {body.model} with an MCP server attached as a tool") def chat_with_mcp(self, headers: AuthHeaders, body: ChatBody) -> ChatResponse: """POST /chat/completions carrying the LiteLLM key in `headers` (either ingress form) with an MCP server attached in `body.tools`. The gateway diff --git a/tests/e2e/mcp/oauth_gateway.py b/tests/e2e/mcp/oauth_gateway.py index 029b0135900..2abe8262b6e 100644 --- a/tests/e2e/mcp/oauth_gateway.py +++ b/tests/e2e/mcp/oauth_gateway.py @@ -21,6 +21,7 @@ from typing import Final import psycopg from e2e_config import INHERITED_ENV_PREFIXES, available_port from e2e_http import NoBody +from e2e_metadata import step from idp import Keycloak, stop_process_group from proxy_client import ProxyClient, build_proxy_client from psycopg.rows import class_row @@ -37,6 +38,7 @@ class CredentialRow: credential_b64: str = field(repr=False) +@step("Read the user's stored OAuth credential for the MCP server from the database and decrypt it") def stored_oauth(user_id: str, server_id: str) -> StoredOAuth: """Read the encrypted credential because management APIs omit the plaintext token.""" from litellm.proxy.common_utils.encrypt_decrypt_utils import decrypt_value_helper @@ -108,6 +110,7 @@ class OAuthGateway: _log_path: Path _child: subprocess.Popen[bytes] | None = field(default=None, init=False, repr=False) + @step("Start the separate LiteLLM proxy and wait for /health/liveliness") def start(self) -> None: with self._log_path.open("ab") as log: self._child = subprocess.Popen( @@ -126,11 +129,13 @@ class OAuthGateway: time.sleep(0.5) raise AssertionError("owned OAuth gateway did not become ready") + @step("Stop the separate LiteLLM proxy") def stop(self) -> None: if self._child is not None: stop_process_group(self._child) assert self._child.poll() is not None, "old gateway process is still alive" + @step("Restart the separate LiteLLM proxy process so its in-memory caches start empty") def restart(self) -> None: assert self._child is not None previous: Final = self._child.pid @@ -139,6 +144,7 @@ class OAuthGateway: assert self._child.pid != previous, "gateway restart did not create a new process" +@step("Start a separate LiteLLM proxy from source with JWT auth against Keycloak") def owned_gateway(idp: Keycloak, directory: Path, cleanup: ExitStack) -> OAuthGateway: for name in ("DATABASE_URL", "LITELLM_LICENSE", "LITELLM_SALT_KEY", "LITELLM_MASTER_KEY"): assert os.environ.get(name), f"{name} is required for the owned OAuth gateway" diff --git a/tests/e2e/mcp/test_mcp_access_group_e2e.py b/tests/e2e/mcp/test_mcp_access_group_e2e.py index f72b75fd43d..744acb78aea 100644 --- a/tests/e2e/mcp/test_mcp_access_group_e2e.py +++ b/tests/e2e/mcp/test_mcp_access_group_e2e.py @@ -16,6 +16,7 @@ import pytest from datadog_mcp import SEARCH_LOGS_TOOL, register_datadog_mcp from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager from mcp_client import McpClient @@ -24,6 +25,12 @@ pytestmark = pytest.mark.e2e class TestMcpAccessGroupToolSelection: @pytest.mark.covers("mcp.list_tools.api_key.access_group_scoped") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_access_group_scopes_tool_selection( self, client: McpClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/mcp/test_mcp_chat_completion_oauth_e2e.py b/tests/e2e/mcp/test_mcp_chat_completion_oauth_e2e.py index 01e94f7b86f..9d7e4713963 100644 --- a/tests/e2e/mcp/test_mcp_chat_completion_oauth_e2e.py +++ b/tests/e2e/mcp/test_mcp_chat_completion_oauth_e2e.py @@ -30,6 +30,7 @@ import pytest from e2e_config import CHEAP_ANTHROPIC_MODEL, LINEAR_MCP_URL, LINEAR_STORAGE_STATE, unique_marker from e2e_http import AuthHeaders +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, KeyGenerateBody, McpChatTool, McpServerCreateBody, ObjectPermission from proxy_client import ProxyClient @@ -70,6 +71,16 @@ class TestMcpChatCompletionOauth: @pytest.mark.covers("mcp.list_tools.oauth.succeeds") @pytest.mark.covers("mcp.call_tool.oauth.succeeds") + @meta( + Subject( + domain=Domain.MCP, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completion_uses_linear_with_x_litellm_api_key_header( self, chat_client: ChatMcpClient, resources: ResourceManager ) -> None: @@ -134,6 +145,16 @@ class TestMcpChatCompletionOauth: @pytest.mark.covers("mcp.list_tools.oauth.succeeds") @pytest.mark.covers("mcp.call_tool.oauth.succeeds") + @meta( + Subject( + domain=Domain.MCP, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + capabilities=(Capability.FUNCTION_CALLING,), + mode=Mode.NONSTREAM, + ) + ) def test_chat_completion_uses_linear_with_authorization_bearer_header( self, chat_client: ChatMcpClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/mcp/test_mcp_datadog_e2e.py b/tests/e2e/mcp/test_mcp_datadog_e2e.py index 031fbf6d936..e383e9ab766 100644 --- a/tests/e2e/mcp/test_mcp_datadog_e2e.py +++ b/tests/e2e/mcp/test_mcp_datadog_e2e.py @@ -13,10 +13,10 @@ from __future__ import annotations import pytest -from conftest import DdLogsReader -from datadog_mcp import SEARCH_LOGS_TOOL, assert_dd_mcp_creds, register_datadog_mcp +from datadog_mcp import SEARCH_LOGS_TOOL, DdLogsReader, assert_dd_mcp_creds, register_datadog_mcp from e2e_config import CHEAP_ANTHROPIC_MODEL, DD_SEARCH_FROM, unique_marker from e2e_http import NoBody, unwrap +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from mcp_client import McpClient from models import ChatBody, ChatMessage @@ -50,6 +50,15 @@ def _seed_completion(proxy: ProxyClient, *, key: str, marker: str) -> None: class TestDatadogMcpRoundTrip: @pytest.mark.covers("mcp.list_tools.api_key.succeeds", "mcp.call_tool.api_key.succeeds") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_ANTHROPIC_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_search_logs_finds_seeded_completion( self, client: McpClient, diff --git a/tests/e2e/mcp/test_mcp_guardrail_e2e.py b/tests/e2e/mcp/test_mcp_guardrail_e2e.py index 92c632cb316..d34bd80e564 100644 --- a/tests/e2e/mcp/test_mcp_guardrail_e2e.py +++ b/tests/e2e/mcp/test_mcp_guardrail_e2e.py @@ -23,6 +23,7 @@ import pytest from datadog_mcp import SEARCH_LOGS_TOOL, assert_dd_mcp_creds, register_datadog_mcp from e2e_config import DD_SEARCH_FROM, unique_marker from e2e_http import Result, Success, UnknownApiError +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager from mcp_client import McpCallToolResponse, McpClient, McpToolArguments @@ -82,6 +83,12 @@ class TestMcpToolCallGuardrail: "guardrail.litellm_content_filter.pre_mcp_call.blocks", exercised_on=["mcp_operations"], ) + @meta( + Subject( + domain=Domain.GUARDRAILS, + route=Route.MCP, + ) + ) def test_content_filter_blocks_banned_keyword_in_tool_args( self, client: McpClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/mcp/test_mcp_key_access_e2e.py b/tests/e2e/mcp/test_mcp_key_access_e2e.py index c00d67bc9cf..9f5e04ccb9b 100644 --- a/tests/e2e/mcp/test_mcp_key_access_e2e.py +++ b/tests/e2e/mcp/test_mcp_key_access_e2e.py @@ -18,6 +18,7 @@ from typing import Final from datadog_mcp import SEARCH_LOGS_TOOL, register_datadog_mcp from e2e_config import DD_SEARCH_FROM, unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager from mcp_client import McpClient from models import KeyGenerateBody, ObjectPermission @@ -33,6 +34,12 @@ def _key(client: McpClient, resources: ResourceManager, *, mcp_servers: list[str class TestMcpKeyGrantByAlias: + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_alias_grant_persists_verbatim_and_lists_tools( self, client: McpClient, @@ -62,6 +69,12 @@ class TestMcpKeyGrantByAlias: class TestMcpKeyWithoutAccessIsDenied: @pytest.mark.covers("mcp.list_tools.api_key.denied_without_permission") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_list_tools_denied_without_permission( self, client: McpClient, @@ -82,6 +95,12 @@ class TestMcpKeyWithoutAccessIsDenied: ) @pytest.mark.covers("mcp.call_tool.api_key.denied_without_permission") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_call_tool_denied_without_permission( self, client: McpClient, @@ -113,6 +132,12 @@ class TestMcpKeyWithoutAccessIsDenied: class TestMcpHealthVisibility: + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_route_restricted_health_matches_server_grants( self, client: McpClient, diff --git a/tests/e2e/mcp/test_mcp_oauth_happy_path_e2e.py b/tests/e2e/mcp/test_mcp_oauth_happy_path_e2e.py index 81470c21d51..b57c941cb18 100644 --- a/tests/e2e/mcp/test_mcp_oauth_happy_path_e2e.py +++ b/tests/e2e/mcp/test_mcp_oauth_happy_path_e2e.py @@ -18,6 +18,7 @@ from typing import Final, Literal import pytest from e2e_config import LINEAR_MCP_URL, LINEAR_READONLY_TOOL, LINEAR_STORAGE_STATE, unique_marker from e2e_http import AuthHeaders, NoBody, get_external, unwrap +from e2e_metadata import Domain, Route, Subject, meta from idp import Identity, Keycloak from lifecycle import ResourceManager from models import ( @@ -88,6 +89,12 @@ class TestMcpOauthHappyPath: @pytest.mark.covers("mcp.list_tools.oauth.succeeds") @pytest.mark.covers("mcp.call_tool.oauth.succeeds") @pytest.mark.covers("mcp.call_tool.oauth.persists_across_processes") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) @pytest.mark.parametrize("route", ("aggregate_sso", "explicit_header_jwt")) @pytest.mark.parametrize("observed", (False, True), ids=("direct", "observed")) def test_consent_list_call_and_cold_restart( diff --git a/tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py b/tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py index 6b901145eb1..cff72f3c95d 100644 --- a/tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py +++ b/tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py @@ -18,6 +18,7 @@ import pytest from datadog_mcp import SEARCH_LOGS_TOOL, register_datadog_mcp from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Domain, Route, Subject, meta from lifecycle import ResourceManager from mcp_client import McpClient from models import ToolsetCreateBody, ToolsetTool @@ -60,6 +61,12 @@ def _wire_prefix(wire_name: str, tool_name: str, catalog: frozenset[str]) -> str class TestMcpToolsetEnforcement: @pytest.mark.covers("mcp.list_tools.api_key.toolset_scoped") + @meta( + Subject( + domain=Domain.MCP, + route=Route.MCP, + ) + ) def test_key_granted_a_toolset_lists_exactly_its_tools(self, client: McpClient, resources: ResourceManager) -> None: server_id: Final = register_datadog_mcp(client, resources, allowed_tools=None) client.await_registered(server_id) diff --git a/tests/e2e/router/reliability_support.py b/tests/e2e/router/reliability_support.py index 04fa30a0d14..03e585273c0 100644 --- a/tests/e2e/router/reliability_support.py +++ b/tests/e2e/router/reliability_support.py @@ -23,6 +23,7 @@ from pydantic import BaseModel, ValidationError from proxy_client import ProxyClient from e2e_config import CHEAP_OPENAI_MODEL, PROXY_BASE_URL, unique_marker from e2e_http import NetworkError, StreamHead, StreamingResponse +from e2e_metadata import step from models import ( CacheControl, ChatMessage, @@ -79,6 +80,7 @@ def cached_system_turn(marker: str) -> ChatMessage: return ChatMessage(role="system", content=[TextContentPart(text=filler, cache_control=CacheControl())]) +@step(f"Add a deployment named {{name}} that calls {REAL_MODEL} at an unreachable address") def create_bad_base_deployment(proxy: ProxyClient, name: str) -> str: """Register a deployment pointing at an unreachable base, so every call to it fails with a real connection error the fallback can reroute around.""" @@ -87,6 +89,7 @@ def create_bad_base_deployment(proxy: ProxyClient, name: str) -> str: ) +@step(f"Add a deployment named {{name}} that calls {REAL_MODEL} at an unreachable address and is never benched") def create_never_benched_refusing_deployment(proxy: ProxyClient, name: str) -> str: return proxy.create_model( name, @@ -94,6 +97,7 @@ def create_never_benched_refusing_deployment(proxy: ProxyClient, name: str) -> s ) +@step(f"Add a deployment named {{name}} that calls {REAL_MODEL} with a 1ms timeout") def create_timeout_deployment(proxy: ProxyClient, name: str) -> str: """Register a deployment with a 1ms deadline the real backend always exceeds.""" return proxy.create_model( @@ -101,12 +105,14 @@ def create_timeout_deployment(proxy: ProxyClient, name: str) -> str: ) +@step(f"Add a deployment named {{name}} that calls the small-context model {SMALL_CONTEXT_MODEL}") def create_small_context_deployment(proxy: ProxyClient, name: str) -> str: """Register a deployment on the smallest-context model OpenAI still serves, so an oversized prompt earns a real context-window refusal from the provider.""" return proxy.create_model(name, LiteLLMParamsBody(model=SMALL_CONTEXT_MODEL, api_key=REAL_KEY)) +@step(f"Add a deployment named {{name}} that calls {AZURE_MODEL} behind Azure's content filter") def create_content_filtered_deployment(proxy: ProxyClient, name: str) -> str: """Register the Azure OpenAI deployment whose content filter refuses CONTENT_POLICY_PROMPT with a real policy-violation 400 (the one live trigger @@ -124,6 +130,10 @@ def create_content_filtered_deployment(proxy: ProxyClient, name: str) -> str: ) +@step( + f"Add a deployment named {{name}} that calls {AZURE_MODEL} and is benched for {{cooldown_time}}s" + " on its first failure" +) def create_azure_benched_on_first_failure_deployment(proxy: ProxyClient, name: str, cooldown_time: float) -> str: """The live Azure OpenAI deployment holding all of the group's shuffle weight, benched on its first failure of any class, with the client's own retries off.""" @@ -144,6 +154,7 @@ def create_azure_benched_on_first_failure_deployment(proxy: ProxyClient, name: s ) +@step(f"Add a deployment named {{name}} that calls {CACHING_MODEL} with prompt caching") def create_caching_deployment(proxy: ProxyClient, name: str) -> str: """Register the Anthropic deployment whose prompt cache the affinity check pins to.""" return proxy.create_model(name, LiteLLMParamsBody(model=CACHING_MODEL, api_key=CACHING_KEY, weight=1)) @@ -165,6 +176,7 @@ def _register_benched_on_first_failure( ) +@step(f"Add a deployment named {{name}} that calls {REAL_MODEL} with a 1ms timeout and is benched on its first timeout") def create_always_timing_out_deployment(proxy: ProxyClient, name: str, cooldown_time: float | None = None) -> str: """A 1ms deadline the real backend always exceeds, benched on its first Timeout.""" return _register_benched_on_first_failure( @@ -176,6 +188,7 @@ def create_always_timing_out_deployment(proxy: ProxyClient, name: str, cooldown_ ) +@step(f"Add a deployment named {{name}} that calls {REAL_MODEL} with an invalid key and is benched on its first 401") def create_always_unauthorized_deployment(proxy: ProxyClient, name: str, cooldown_time: float | None = None) -> str: """A key the real backend rejects with a 401, benched on its first AuthenticationError.""" return _register_benched_on_first_failure( @@ -204,6 +217,10 @@ def _nested_proxy_params(upstream_group: str, upstream_key: str, cooldown_time: ) +@step( + "Add a deployment named {name} that fronts {upstream_group} on this proxy, so it always gets a 500" + " and is benched on the first one" +) def create_always_5xx_deployment( proxy: ProxyClient, name: str, upstream_group: str, upstream_key: str, cooldown_time: float | None = None ) -> str: @@ -217,6 +234,10 @@ def create_always_5xx_deployment( ) +@step( + "Add a deployment named {name} that fronts {upstream_group} on this proxy with a key out of rpm," + " so it always gets a 429 and is benched on the first one" +) def create_always_rate_limited_deployment( proxy: ProxyClient, name: str, upstream_group: str, upstream_key: str, cooldown_time: float | None = None ) -> str: @@ -227,6 +248,7 @@ def create_always_rate_limited_deployment( ) +@step(f"Use up the rpm-limited key's one allowed request with a /chat/completions call to {CHEAP_OPENAI_MODEL}") def spend_only_request_of(proxy: ProxyClient, spent_key: str) -> None: """Uses up the one request an rpm_limit=1 key allows. The proxy's rate limiter opens the key's 60s window on this call, so it goes right before the calls that @@ -239,6 +261,7 @@ def spend_only_request_of(proxy: ProxyClient, spent_key: str) -> None: ) +@step(f"Add a deployment named {{name}} that calls {SMALL_CONTEXT_MODEL} and takes all of its group's traffic") def create_always_picked_small_context_deployment(proxy: ProxyClient, name: str) -> str: """The always-picked half of a retry pair on the smallest-context model OpenAI still serves: it holds all of the model group's shuffle weight, so an oversized @@ -253,12 +276,14 @@ def create_always_picked_small_context_deployment(proxy: ProxyClient, name: str) ) +@step(f"Add a deployment named {{name}} for {REAL_MODEL} that answers with a canned reply") def create_canned_deployment(proxy: ProxyClient, name: str) -> str: """A deployment that answers from a canned reply, so a call to it goes through the router's deployment pick like any other but never reaches a provider.""" return proxy.create_model(name, LiteLLMParamsBody(model=REAL_MODEL, mock_response="ok")) +@step(f"Add a zero-weight backup deployment named {{name}} that calls {REAL_MODEL}") def create_zero_weight_backup_deployment(proxy: ProxyClient, name: str) -> str: """The other half of a retry pair: healthy, but weight 0, so the weighted shuffle never opens on it. It is reachable only once its sibling is out of the running, @@ -273,6 +298,7 @@ def create_zero_weight_backup_deployment(proxy: ProxyClient, name: str) -> str: ) +@step("Send a /chat/completions request to {model} with a full message history and stream set to {stream}") def chat_turns_override( proxy: ProxyClient, key: str, @@ -300,6 +326,7 @@ def chat_turns_override( ) +@step("Send a /chat/completions request to {model} with stream set to {stream}") def chat_override( proxy: ProxyClient, key: str, @@ -322,6 +349,7 @@ def chat_override( ) +@step('Open a streaming /chat/completions request to {model} with the prompt "{content}" and leave it in flight') def open_chat_stream( proxy: ProxyClient, key: str, diff --git a/tests/e2e/router/test_auto_router_regressions_e2e.py b/tests/e2e/router/test_auto_router_regressions_e2e.py index 374badcf5fc..898b164e713 100644 --- a/tests/e2e/router/test_auto_router_regressions_e2e.py +++ b/tests/e2e/router/test_auto_router_regressions_e2e.py @@ -49,6 +49,7 @@ import pytest from pydantic import BaseModel, ConfigDict, Field from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta from e2e_http import AnthropicHeaders, AuthHeaders, UnauthorizedError, unwrap from lifecycle import ResourceManager from models import ( @@ -341,6 +342,15 @@ def credentialed_alias(proxy: ProxyClient, router_stack: ExitStack) -> Credentia class TestTagSplitRouting: @pytest.mark.covers("reliability.routing.tagged_marker.request_tag_selects_marker") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_body_tagged_chat_routes_through_the_marker_to_its_tier( self, proxy: ProxyClient, resources: ResourceManager, plain_first_split: TagSplitDeployment ) -> None: @@ -355,6 +365,15 @@ class TestTagSplitRouting: _assert_served_only_by(rows, CHEAP_SERVED | {plain_first_split.tier}, "body-tagged chat on the shared name") @pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(PLAIN_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_untagged_chat_is_always_served_by_the_plain_deployment( self, proxy: ProxyClient, resources: ResourceManager, plain_first_split: TagSplitDeployment ) -> None: @@ -370,6 +389,15 @@ class TestTagSplitRouting: _assert_served_only_by(rows, PLAIN_SERVED | {plain_first_split.shared}, "untagged chat on the shared name") @pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(PLAIN_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_untagged_messages_is_served_by_the_plain_deployment( self, proxy: ProxyClient, resources: ResourceManager, plain_first_split: TagSplitDeployment ) -> None: @@ -387,6 +415,15 @@ class TestTagSplitRouting: class TestUntaggedTierDeployments: @pytest.mark.covers("reliability.routing.tagged_marker.header_tag_selects_marker") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.MESSAGES, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_header_tagged_messages_routes_through_the_marker_to_an_untagged_tier( self, proxy: ProxyClient, resources: ResourceManager, marker_first_split: TagSplitDeployment ) -> None: @@ -413,6 +450,15 @@ class TestUntaggedTierDeployments: ) @pytest.mark.covers("reliability.routing.tagged_marker.untagged_tier_deployments_still_served") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_body_tagged_chat_reaches_the_untagged_tier_after_marker_rewrite( self, proxy: ProxyClient, resources: ResourceManager, marker_first_split: TagSplitDeployment ) -> None: @@ -431,6 +477,12 @@ class TestUntaggedTierDeployments: _assert_served_only_by(rows, CHEAP_SERVED | {marker_first_split.tier}, "body-tagged chat with untagged tier") @pytest.mark.covers("reliability.routing.tagged_marker.tag_semantics_stay_strict") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.CHAT_COMPLETIONS, + ) + ) def test_tagged_call_straight_at_an_untagged_deployment_stays_denied( self, proxy: ProxyClient, resources: ResourceManager, marker_first_split: TagSplitDeployment ) -> None: @@ -449,6 +501,15 @@ class TestUntaggedTierDeployments: class TestResponsesApiTagRouting: @pytest.mark.covers("reliability.routing.tagged_marker.responses_input_routes_through_marker") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.RESPONSES, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_header_tagged_responses_with_string_input_routes_to_the_tier( self, proxy: ProxyClient, resources: ResourceManager, plain_first_split: TagSplitDeployment ) -> None: @@ -471,6 +532,15 @@ class TestResponsesApiTagRouting: ) @pytest.mark.covers("reliability.routing.tagged_marker.responses_input_routes_through_marker") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.RESPONSES, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_body_tagged_responses_with_list_input_routes_to_the_tier( self, proxy: ProxyClient, resources: ResourceManager, plain_first_split: TagSplitDeployment ) -> None: @@ -497,6 +567,15 @@ class TestResponsesApiTagRouting: _assert_served_only_by(rows, CHEAP_SERVED | {plain_first_split.tier}, "body-tagged /v1/responses list input") @pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.RESPONSES, + providers=(Provider.ANTHROPIC,), + models=(PLAIN_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_untagged_responses_is_served_by_the_plain_deployment( self, proxy: ProxyClient, resources: ResourceManager, plain_first_split: TagSplitDeployment ) -> None: @@ -524,6 +603,14 @@ class TestResponsesApiTagRouting: class TestStrategyAliasPricing: @pytest.mark.covers("reliability.routing.strategy_alias.custom_pricing_ignored") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_zero_priced_alias_still_logs_spend_at_the_tier_rate( self, proxy: ProxyClient, resources: ResourceManager, zero_priced_alias: ZeroPricedAlias ) -> None: @@ -546,6 +633,14 @@ class TestStrategyAliasPricing: class TestComplexityHeuristicScope: @pytest.mark.covers("reliability.routing.complexity_heuristic.scores_current_ask_only") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_trivial_ask_behind_keyword_heavy_system_prompt_stays_on_the_cheap_tier( self, proxy: ProxyClient, resources: ResourceManager, heuristic_split: HeuristicSplit ) -> None: @@ -573,6 +668,15 @@ class TestComplexityHeuristicScope: class TestSemanticAutoRouterResponses: @pytest.mark.covers("reliability.routing.semantic_auto_router.responses_input_routed") + @meta( + Subject( + domain=Domain.ROUTING, + route=Route.RESPONSES, + providers=(Provider.ANTHROPIC, Provider.OPENAI,), + models=(CHEAP_MODEL, EMBEDDING_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_responses_input_reaches_the_semantic_auto_router( self, proxy: ProxyClient, resources: ResourceManager, semantic_auto_router: SemanticAutoRouter ) -> None: @@ -617,6 +721,14 @@ class TestSemanticAutoRouterResponses: class TestAliasParamForwarding: @pytest.mark.covers("reliability.routing.tagged_marker.alias_connection_params_stay_with_tier") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.ANTHROPIC,), + models=(CHEAP_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_alias_api_key_never_overrides_the_tier_credential( self, proxy: ProxyClient, resources: ResourceManager, credentialed_alias: CredentialedAlias ) -> None: diff --git a/tests/e2e/router/test_complexity_router_e2e.py b/tests/e2e/router/test_complexity_router_e2e.py index e8508c963b8..c8f20102223 100644 --- a/tests/e2e/router/test_complexity_router_e2e.py +++ b/tests/e2e/router/test_complexity_router_e2e.py @@ -19,9 +19,12 @@ anthropic proves the classifier ran and openai proves it silently fell back - th exact failure before the fix. """ +from typing import Final + import pytest from complexity_router_client import ComplexityRouterClient +from e2e_metadata import Domain, Mode, Provider, Subject, meta from e2e_http import unwrap from models import ChatBody, ChatMessage @@ -33,9 +36,11 @@ LEXICALLY_SIMPLE_HARD_PROMPT = "Should I pay off my mortgage early or invest the # SIMPLE tier backend; served only when the classifier silently falls back to heuristic. # Spend logs may store the alias (gpt-5.5) or the provider-prefixed form depending on # how the deployment is registered (compose vs /model/new). -HEURISTIC_TIER_MODELS = frozenset({"openai/gpt-5.5", "gpt-5.5"}) +HEURISTIC_TIER_BACKEND: Final = "openai/gpt-5.5" +HEURISTIC_TIER_MODELS = frozenset({HEURISTIC_TIER_BACKEND, "gpt-5.5"}) # MEDIUM/COMPLEX/REASONING tier backend; served only when the LLM classifier runs. -LLM_TIER_MODELS = frozenset({"anthropic/claude-haiku-4-5", "claude-haiku-4-5"}) +LLM_TIER_BACKEND: Final = "anthropic/claude-haiku-4-5" +LLM_TIER_MODELS = frozenset({LLM_TIER_BACKEND, "claude-haiku-4-5"}) @pytest.mark.usefixtures("_ensure_complexity_smart_router") @@ -45,6 +50,14 @@ class TestComplexityRouterLlmClassifier: "(e.g. Is P equal to NP?); re-enable when classifier tier quality is fixed" ) @pytest.mark.covers("reliability.routing.complexity_llm_classifier.routes_by_llm_tier") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI, Provider.ANTHROPIC), + models=(HEURISTIC_TIER_BACKEND, LLM_TIER_BACKEND), + mode=Mode.NONSTREAM, + ) + ) def test_llm_classifier_runs_and_routes_by_semantic_tier( self, client: ComplexityRouterClient, complexity_key: str ) -> None: diff --git a/tests/e2e/router/test_reliability_cache_e2e.py b/tests/e2e/router/test_reliability_cache_e2e.py index f7a2f2ffeb7..5452853a57e 100644 --- a/tests/e2e/router/test_reliability_cache_e2e.py +++ b/tests/e2e/router/test_reliability_cache_e2e.py @@ -18,6 +18,7 @@ from e2e_config import ( REQUEST_TIMEOUT, unique_marker, ) +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody from provider_edge import ProviderRequestObservation, observed_provider_edge @@ -25,6 +26,8 @@ from pydantic import BaseModel, JsonValue pytestmark = [pytest.mark.e2e, pytest.mark.replayable] +CACHE_MODEL: Final = "openai/gpt-5.6" + class _CacheChatBody(ChatBody): ttl: int = 600 @@ -38,6 +41,14 @@ class _CachedAnswer(BaseModel): class TestReliabilityCache: @pytest.mark.covers("reliability.cache.exact.returns_cached") + @meta( + Subject( + domain=Domain.CACHING, + providers=(Provider.OPENAI,), + models=(CACHE_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_exact_cache_returns_cached( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -57,7 +68,7 @@ class TestReliabilityCache: model_id: Final = client.proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-5.6", + model=CACHE_MODEL, api_key="os.environ/OPENAI_API_KEY", api_base=f"{edge.api_base('openai')}/v1", ), diff --git a/tests/e2e/router/test_reliability_cancel_on_disconnect_e2e.py b/tests/e2e/router/test_reliability_cancel_on_disconnect_e2e.py index 06174e97d20..e39f1ddcb90 100644 --- a/tests/e2e/router/test_reliability_cancel_on_disconnect_e2e.py +++ b/tests/e2e/router/test_reliability_cancel_on_disconnect_e2e.py @@ -29,9 +29,11 @@ import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker from e2e_http import AbandonedRequest, StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ChatMessage, ReliabilityChatBody, RouterSettingsOverride from reliability_support import ( + AZURE_MODEL, REPLICA_PROPAGATION_SECONDS, chat_override, create_azure_benched_on_first_failure_deployment, @@ -102,6 +104,14 @@ def _hang_up_mid_answer(client: ComplexityRouterClient, key: str, group: str) -> class TestReliabilityCancelOnDisconnect: @pytest.mark.covers("reliability.cooldown.client_disconnect.stays_healthy") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.AZURE,), + models=(AZURE_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_client_hanging_up_never_benches_the_deployment( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/router/test_reliability_cooldowns_e2e.py b/tests/e2e/router/test_reliability_cooldowns_e2e.py index 2456bfb5f85..bdb02256976 100644 --- a/tests/e2e/router/test_reliability_cooldowns_e2e.py +++ b/tests/e2e/router/test_reliability_cooldowns_e2e.py @@ -71,10 +71,12 @@ import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import CHEAP_OPENAI_MODEL, unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import KeyGenerateBody, RouterSettingsOverride from reliability_support import ( COOLDOWN_SECONDS, + REAL_MODEL, REPLICA_PROPAGATION_SECONDS, chat_override, create_always_5xx_deployment, @@ -212,6 +214,14 @@ def _assert_trips_then_recovers( class TestReliabilityCooldowns: @pytest.mark.covers("reliability.cooldown.5xx.trips_then_recovers") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_5xx_trips_cooldown_then_recovers( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -230,6 +240,14 @@ class TestReliabilityCooldowns: _assert_trips_then_recovers(client, scoped_key, group, failing, backup, failure_status=500) @pytest.mark.covers("reliability.cooldown.sibling_replica.serves_backup_within_read_interval") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_sibling_replica_serves_backup_within_redis_read_interval( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -272,6 +290,14 @@ class TestReliabilityCooldowns: ) @pytest.mark.covers("reliability.cooldown.429.trips_then_recovers") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL, REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_429_trips_cooldown_then_recovers( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -292,6 +318,14 @@ class TestReliabilityCooldowns: _assert_trips_then_recovers(client, scoped_key, group, failing, backup, failure_status=429) @pytest.mark.covers("reliability.cooldown.auth.trips_then_recovers") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_auth_failure_trips_cooldown_then_recovers( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -304,6 +338,14 @@ class TestReliabilityCooldowns: _assert_trips_then_recovers(client, scoped_key, group, failing, backup, failure_status=401) @pytest.mark.covers("reliability.cooldown.timeout.trips_then_recovers") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_timeout_trips_cooldown_then_recovers( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/router/test_reliability_fallbacks_e2e.py b/tests/e2e/router/test_reliability_fallbacks_e2e.py index 54c11b163d0..61115e00d07 100644 --- a/tests/e2e/router/test_reliability_fallbacks_e2e.py +++ b/tests/e2e/router/test_reliability_fallbacks_e2e.py @@ -31,10 +31,14 @@ import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import RouterSettingsOverride from reliability_support import ( + AZURE_MODEL, CONTENT_POLICY_PROMPT, + REAL_MODEL, + SMALL_CONTEXT_MODEL, azure_prompt_filter_skipped, chat_override, completion_tokens_of, @@ -50,6 +54,8 @@ from reliability_support import ( pytestmark = pytest.mark.e2e +FALLBACK_MODEL: Final = "gpt-5.5" + def _assert_served_by_fallback(resp: StreamingResponse) -> None: assert resp.status_code == 200, f"expected 200 after fallback, got {resp.status_code}: {resp.body[:300]}" @@ -95,6 +101,14 @@ def _filter_verdict(resp: StreamingResponse) -> str: class TestReliabilityFallbacks: @pytest.mark.covers("reliability.fallback.5xx.routes_to_fallback") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(FALLBACK_MODEL, REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_5xx_routes_to_fallback( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -107,11 +121,19 @@ class TestReliabilityFallbacks: scoped_key, primary, f"say hi {unique_marker()}", - override=RouterSettingsOverride(fallbacks=[{primary: ["gpt-5.5"]}]), + override=RouterSettingsOverride(fallbacks=[{primary: [FALLBACK_MODEL]}]), ) _assert_served_by_fallback(resp) @pytest.mark.covers("reliability.fallback.timeout.routes_to_fallback") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(FALLBACK_MODEL, REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_timeout_routes_to_fallback( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -124,11 +146,19 @@ class TestReliabilityFallbacks: scoped_key, primary, f"say hi {unique_marker()}", - override=RouterSettingsOverride(fallbacks=[{primary: ["gpt-5.5"]}]), + override=RouterSettingsOverride(fallbacks=[{primary: [FALLBACK_MODEL]}]), ) _assert_served_by_fallback(resp) @pytest.mark.covers("reliability.fallback.context_window.routes_to_fallback") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(FALLBACK_MODEL, SMALL_CONTEXT_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_context_window_routes_to_fallback( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -141,11 +171,19 @@ class TestReliabilityFallbacks: scoped_key, primary, oversized_prompt(unique_marker()), - override=RouterSettingsOverride(context_window_fallbacks=[{primary: ["gpt-5.5"]}]), + override=RouterSettingsOverride(context_window_fallbacks=[{primary: [FALLBACK_MODEL]}]), ) _assert_served_by_fallback(resp) @pytest.mark.covers("reliability.fallback.content_policy.routes_to_fallback") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.AZURE, Provider.OPENAI,), + models=(AZURE_MODEL, FALLBACK_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_content_policy_routes_to_fallback( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -167,7 +205,7 @@ class TestReliabilityFallbacks: scoped_key, primary, f"{CONTENT_POLICY_PROMPT} {unique_marker()}", - override=RouterSettingsOverride(content_policy_fallbacks=[{primary: ["gpt-5.5"]}]), + override=RouterSettingsOverride(content_policy_fallbacks=[{primary: [FALLBACK_MODEL]}]), ) ) _assert_served_by_fallback(resp) diff --git a/tests/e2e/router/test_reliability_memory_e2e.py b/tests/e2e/router/test_reliability_memory_e2e.py index 77d3a68cae5..9ae6edb3eb2 100644 --- a/tests/e2e/router/test_reliability_memory_e2e.py +++ b/tests/e2e/router/test_reliability_memory_e2e.py @@ -80,11 +80,12 @@ from e2e_config import ( PROXY_REPLICA_URLS, unique_marker, ) +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from memory_readings import RssCapture, RssReading, WorkerKey, read_rss_everywhere from models import ChatMessage, RouterSettingsOverride, SpendLogRow from proxy_client import ProxyClient -from reliability_support import chat_override, create_never_benched_refusing_deployment +from reliability_support import REAL_MODEL, chat_override, create_never_benched_refusing_deployment pytestmark = [pytest.mark.e2e, pytest.mark.quiet_stack] @@ -222,6 +223,11 @@ def _stored_request_kb(proxy: ProxyClient, call: FailedCall) -> float: class TestReliabilityMemory: @pytest.mark.covers("reliability.perf.idle_memory.under_slo") + @meta( + Subject( + domain=Domain.DEPLOY_OPS, + ) + ) def test_workers_idle_under_rss_budget_before_traffic(self, idle_rss: RssCapture) -> None: assert not idle_rss.failures, ( f"{len(idle_rss.failures)} replica(s) gave no RSS reading when the session started, so their idle " @@ -242,6 +248,14 @@ class TestReliabilityMemory: ) @pytest.mark.covers("reliability.perf.memory.under_slo") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_failing_requests_do_not_grow_rss_or_stored_request( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/router/test_reliability_prompt_caching_e2e.py b/tests/e2e/router/test_reliability_prompt_caching_e2e.py index 667b398cd16..7bf23bf967f 100644 --- a/tests/e2e/router/test_reliability_prompt_caching_e2e.py +++ b/tests/e2e/router/test_reliability_prompt_caching_e2e.py @@ -23,9 +23,11 @@ import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker +from e2e_metadata import Capability, Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ChatMessage, LiteLLMParamsBody, ModelInfoBody, ModelNewBody from reliability_support import ( + CACHING_MODEL, REAL_KEY, REAL_MODEL, cached_system_turn, @@ -42,6 +44,15 @@ FOLLOW_UPS = 3 class TestReliabilityPromptCachingAffinity: @pytest.mark.covers("reliability.cache.prompt_caching_model_select.returns_cached") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.ANTHROPIC, Provider.OPENAI), + models=(CACHING_MODEL, REAL_MODEL), + capabilities=(Capability.PROMPT_CACHING,), + mode=Mode.NONSTREAM, + ) + ) def test_cached_conversation_stays_on_deployment_holding_its_cache( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/router/test_reliability_retries_e2e.py b/tests/e2e/router/test_reliability_retries_e2e.py index a90efa52b5f..c6ae8c457e4 100644 --- a/tests/e2e/router/test_reliability_retries_e2e.py +++ b/tests/e2e/router/test_reliability_retries_e2e.py @@ -30,9 +30,12 @@ import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import CHEAP_OPENAI_MODEL, unique_marker from e2e_http import StreamingResponse +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import KeyGenerateBody, RouterSettingsOverride from reliability_support import ( + REAL_MODEL, + SMALL_CONTEXT_MODEL, chat_override, completion_tokens_of, content_of, @@ -84,6 +87,14 @@ def _retry_once(client: ComplexityRouterClient, key: str, group: str) -> Streami class TestReliabilityRetries: @pytest.mark.covers("reliability.retry.timeout.succeeds_within_retries") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_timeout_on_first_deployment_succeeds_on_retry( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -96,6 +107,14 @@ class TestReliabilityRetries: _assert_served_after_retry(_retry_once(client, scoped_key, group)) @pytest.mark.covers("reliability.retry.5xx.succeeds_within_retries") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_5xx_on_first_deployment_succeeds_on_retry( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -112,6 +131,14 @@ class TestReliabilityRetries: _assert_served_after_retry(_retry_once(client, scoped_key, group)) @pytest.mark.covers("reliability.retry.429.succeeds_within_retries") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(CHEAP_OPENAI_MODEL, REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_429_on_first_deployment_succeeds_on_retry( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -130,6 +157,14 @@ class TestReliabilityRetries: _assert_served_after_retry(_retry_once(client, scoped_key, group)) @pytest.mark.covers("reliability.retry.auth.succeeds_within_retries") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_auth_failure_on_first_deployment_succeeds_on_retry( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -142,6 +177,14 @@ class TestReliabilityRetries: _assert_served_after_retry(_retry_once(client, scoped_key, group)) @pytest.mark.covers("reliability.retry.context_window.succeeds_within_retries") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL, SMALL_CONTEXT_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_context_window_refusal_on_first_deployment_succeeds_on_retry( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/router/test_reliability_routing_strategies_e2e.py b/tests/e2e/router/test_reliability_routing_strategies_e2e.py index 2abc2ee5f54..ee0334f989d 100644 --- a/tests/e2e/router/test_reliability_routing_strategies_e2e.py +++ b/tests/e2e/router/test_reliability_routing_strategies_e2e.py @@ -63,6 +63,7 @@ import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker from e2e_http import StreamChunk, StreamHead, StreamStep, StreamTruncation +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import LiteLLMParamsBody, ModelInfoBody, ModelNewBody, RouterSettingsOverride, RoutingStrategy from reliability_support import REAL_KEY, REAL_MODEL, chat_override, model_id_of, open_chat_stream @@ -173,6 +174,14 @@ def _assert_shuffle_control_lands_on(client: ComplexityRouterClient, key: str, g class TestReliabilityRoutingStrategies: @pytest.mark.covers("reliability.routing.simple_shuffle.picks_healthy_deployment") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_simple_shuffle_honors_weights( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -191,6 +200,14 @@ class TestReliabilityRoutingStrategies: ) @pytest.mark.covers("reliability.routing.cost_based.picks_lowest_cost") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_cost_based_picks_cheapest_deployment( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -206,6 +223,14 @@ class TestReliabilityRoutingStrategies: _assert_shuffle_control_lands_on(client, scoped_key, group, pricey) @pytest.mark.covers("reliability.routing.usage_based.picks_under_tpm") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_usage_based_picks_deployment_with_tpm_headroom( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -223,6 +248,14 @@ class TestReliabilityRoutingStrategies: "so latency-based has no signal to route on" ) @pytest.mark.covers("reliability.routing.latency_based.picks_lowest_latency") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + mode=Mode.NONSTREAM, + ) + ) def test_latency_based_routes_around_deployment_that_times_out( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -253,6 +286,13 @@ class TestReliabilityRoutingStrategies: "so least-busy has no signal to route on" ) @pytest.mark.covers("reliability.routing.least_busy.picks_lowest_traffic") + @meta( + Subject( + domain=Domain.ROUTING, + providers=(Provider.OPENAI,), + models=(REAL_MODEL,), + ) + ) def test_least_busy_avoids_deployment_with_request_in_flight( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/router/test_reliability_timeouts_e2e.py b/tests/e2e/router/test_reliability_timeouts_e2e.py index f24d5139e66..926d8539d6d 100644 --- a/tests/e2e/router/test_reliability_timeouts_e2e.py +++ b/tests/e2e/router/test_reliability_timeouts_e2e.py @@ -13,14 +13,16 @@ import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager -from reliability_support import chat_override, create_timeout_deployment +from reliability_support import REAL_MODEL, chat_override, create_timeout_deployment pytestmark = pytest.mark.e2e class TestReliabilityTimeouts: @pytest.mark.covers("reliability.timeout.request_timeout.exceeds_deadline") + @meta(Subject(domain=Domain.ROUTING, providers=(Provider.OPENAI,), models=(REAL_MODEL,), mode=Mode.NONSTREAM)) def test_request_timeout_exceeds_deadline( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -35,6 +37,7 @@ class TestReliabilityTimeouts: assert "timeout" in resp.body.lower(), f"the 408 body should name the timeout, got: {resp.body[:300]}" @pytest.mark.covers("reliability.timeout.stream_timeout.exceeds_deadline") + @meta(Subject(domain=Domain.ROUTING, providers=(Provider.OPENAI,), models=(REAL_MODEL,), mode=Mode.STREAM)) def test_stream_timeout_exceeds_deadline( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: From 3538e87e45d147102b0f1dee034bb9424f966874 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Wed, 7 Oct 2026 10:33:12 -0700 Subject: [PATCH 08/13] test(e2e): tag the remaining quota_management tests and record budget and spend client steps (#44966) * test(e2e): add enum values, auto-discovering label gates and secret hiding for e2e metadata * test(e2e): tag quota_management tests with Subject metadata and record budget client steps * docs(e2e): name every markerless harness test file that carries no Subject * test(e2e): keep the step discovery comprehensions to one for clause --- .../quota_management/budgets/budget_client.py | 27 +++++++++++++ .../test_model_group_alias_rate_limit_e2e.py | 17 +++++++++ .../spend_tracking/cost_rows.py | 4 ++ .../spend_tracking/spend_e2e_client.py | 38 +++++++++++++++++++ .../spend_tracking/spend_reconciliation.py | 6 +++ .../test_service_tier_pricing_e2e.py | 38 ++++++++++++++++++- .../spend_tracking/test_spend_routes.py | 6 +++ .../test_spend_surface_consistency_e2e.py | 12 +++++- .../spend_tracking/test_spend_tracking_e2e.py | 28 +++++++++++++- ...test_websearch_interception_session_e2e.py | 11 ++++++ 10 files changed, 184 insertions(+), 3 deletions(-) diff --git a/tests/e2e/quota_management/budgets/budget_client.py b/tests/e2e/quota_management/budgets/budget_client.py index 087dc8ca522..926dae0d954 100644 --- a/tests/e2e/quota_management/budgets/budget_client.py +++ b/tests/e2e/quota_management/budgets/budget_client.py @@ -17,6 +17,7 @@ from datetime import datetime from pydantic import AliasPath, BaseModel, Field, RootModel from e2e_http import NoBody, Result, StreamingResponse, Success, unwrap +from e2e_metadata import step from proxy_client import ProxyClient from models import ( AnthropicMessagesBody, @@ -253,15 +254,18 @@ class BudgetClient: ) ) + @step("Delete the virtual key") def delete_key(self, key: str) -> None: self.proxy.delete_key(key) + @step("Read the key's budget windows from /key/info") def key_budget_windows(self, key: str) -> list[BudgetWindowState]: """A key's budget_limits windows as /key/info stores them. Each window's reset_at is advanced by the reset job in the same pass that zeroes the window's spend counter, so a strictly-later value proves the wipe ran.""" return self.proxy.key_info(key).budget_limits or [] + @step("Read the team's budget windows from /team/info") def team_budget_windows(self, team_id: str) -> list[BudgetWindowState]: """Team analog of key_budget_windows, read from /team/info.""" match self._team_info(team_id): @@ -270,11 +274,13 @@ class BudgetClient: case _: return [] + @step("Delete the end users {user_ids}") def delete_customers(self, user_ids: list[str]) -> None: self.proxy.delete_customers(user_ids) # ---- chat (raw HTTP outcome: a budget block surfaces as a non-2xx) -- + @step('Send a /chat/completions request to {model} with the prompt "{content}"') def chat( self, key: str, @@ -297,6 +303,7 @@ class BudgetClient: ), ) + @step('Send a /v1/messages request to {model} with the prompt "{content}"') def messages( self, key: str, @@ -317,6 +324,7 @@ class BudgetClient: # ---- internal user -------------------------------------------------- + @step("Create an internal user with max budget: {max_budget}") def create_user(self, *, max_budget: float, budget_duration: str | None = None) -> str: return unwrap( self.proxy.transport.post( @@ -327,6 +335,7 @@ class BudgetClient: ) ).user_id + @step("Delete the internal user") def delete_user(self, user_id: str) -> None: _ = self.proxy.transport.post( "/user/delete", @@ -335,6 +344,7 @@ class BudgetClient: response_type=NoBody, ) + @step("Read the internal user's spend and budget from /user/info") def user_info(self, user_id: str) -> UserInfoRow | None: result = self.proxy.transport.get( "/user/info", @@ -350,6 +360,7 @@ class BudgetClient: # ---- customer / end-user ------------------------------------------- + @step("Create the end user {customer_id}") def create_customer( self, customer_id: str, @@ -369,6 +380,7 @@ class BudgetClient: # ---- organization --------------------------------------------------- + @step("Create the organization {alias} with max budget: {max_budget}") def create_org(self, *, max_budget: float, alias: str, budget_duration: str | None = None) -> str: return unwrap( self.proxy.transport.post( @@ -383,6 +395,7 @@ class BudgetClient: ) ).organization_id + @step("Read the organization's budget id from /organization/info") def org_budget_id(self, org_id: str) -> str | None: """The id of the budget row backing an org; its budget_reset_at is read via budget_info (LIT-4570: /organization/new stores budget_duration without @@ -399,6 +412,7 @@ class BudgetClient: case _: return None + @step("Delete the organization") def delete_org(self, org_id: str) -> None: _ = self.proxy.transport.delete( "/organization/delete", @@ -409,6 +423,7 @@ class BudgetClient: # ---- team ----------------------------------------------------------- + @step("Create the team {alias} and wait until /team/info returns it") def create_team( self, *, @@ -435,6 +450,7 @@ class BudgetClient: self._wait_for_team(team_id) return team_id + @step("Delete the team") def delete_team(self, team_id: str) -> None: _ = self.proxy.transport.post( "/team/delete", @@ -463,6 +479,7 @@ class BudgetClient: assert last is not None raise AssertionError(last) + @step("Add the internal user to the team") def add_team_member(self, team_id: str, user_id: str, *, max_budget_in_team: float | None = None) -> None: last_body = "" for attempt in range(_TEAM_READY_ATTEMPTS): @@ -484,6 +501,7 @@ class BudgetClient: break raise AssertionError(last_body) + @step("Update the team member's budget with /team/member_update") def update_team_member( self, team_id: str, @@ -504,6 +522,7 @@ class BudgetClient: ) assert resp.ok, resp.body + @step("Read the team member's budget reset time from /team/info") def member_budget_reset_at(self, team_id: str, user_id: str) -> str | None: """The member's per-team budget_reset_at as /team/info reports it, or None if no reset is scheduled. The reset job advances this each time the window @@ -519,6 +538,7 @@ class BudgetClient: # ---- tag ------------------------------------------------------------ + @step("Create the tag {name} with max budget: {max_budget}") def create_tag(self, name: str, *, max_budget: float) -> str: resp = self.proxy.transport.send( "/tag/new", @@ -528,6 +548,7 @@ class BudgetClient: assert resp.ok, resp.body return name + @step("Delete the tag {name}") def delete_tag(self, name: str) -> None: _ = self.proxy.transport.post( "/tag/delete", @@ -538,6 +559,7 @@ class BudgetClient: # ---- model access group --------------------------------------------- + @step("Set a shared budget on the model access group {access_group}") def set_access_group_budget( self, access_group: str, @@ -561,6 +583,7 @@ class BudgetClient: ) ) + @step("Read the budget and spend of the model access group {access_group}") def access_group_budget(self, access_group: str) -> AccessGroupBudgetResponse: return unwrap( self.proxy.transport.get( @@ -571,6 +594,7 @@ class BudgetClient: ) ) + @step("Delete the budget on the model access group {access_group}") def delete_access_group_budget(self, access_group: str) -> None: _ = self.proxy.transport.delete( f"/access_group/{access_group}/budget", @@ -581,6 +605,7 @@ class BudgetClient: # ---- budget table --------------------------------------------------- + @step("Create a budget with /budget/new") def create_budget( self, *, @@ -603,6 +628,7 @@ class BudgetClient: ) ).budget_id + @step("Delete the budget") def delete_budget(self, budget_id: str) -> None: _ = self.proxy.transport.post( "/budget/delete", @@ -611,6 +637,7 @@ class BudgetClient: response_type=NoBody, ) + @step("Read the budget from /budget/info") def budget_info(self, budget_id: str) -> tuple[BudgetRow, ...]: result = self.proxy.transport.post( "/budget/info", diff --git a/tests/e2e/quota_management/ratelimit/test_model_group_alias_rate_limit_e2e.py b/tests/e2e/quota_management/ratelimit/test_model_group_alias_rate_limit_e2e.py index 1c3fff47b78..523436f8d1e 100644 --- a/tests/e2e/quota_management/ratelimit/test_model_group_alias_rate_limit_e2e.py +++ b/tests/e2e/quota_management/ratelimit/test_model_group_alias_rate_limit_e2e.py @@ -21,6 +21,7 @@ import time import pytest from e2e_config import unique_marker from e2e_http import StreamingResponse, require_successful_call +from e2e_metadata import Domain, Mode, Provider, Subject, meta from quota_client import QuotaClient pytestmark = pytest.mark.e2e @@ -75,11 +76,27 @@ def _assert_blocked_inside_window( class TestModelGroupAliasRateLimit: @pytest.mark.covers("quota_management.ratelimit.model_group_alias.shares_bucket") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL_GROUP, MODEL_ALIAS), + mode=Mode.NONSTREAM, + ) + ) def test_alias_shares_rpm_bucket_with_model_group(self, client: QuotaClient, scoped_key: str) -> None: opened_at = _exhaust_rpm(client, scoped_key, MODEL_GROUP) _assert_blocked_inside_window(client, scoped_key, MODEL_ALIAS, opened_at) @pytest.mark.covers("quota_management.ratelimit.model_group_alias.shares_bucket") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC,), + models=(MODEL_GROUP, MODEL_ALIAS), + mode=Mode.NONSTREAM, + ) + ) def test_model_group_shares_rpm_bucket_with_alias(self, client: QuotaClient, scoped_key: str) -> None: opened_at = _exhaust_rpm(client, scoped_key, MODEL_ALIAS) _assert_blocked_inside_window(client, scoped_key, MODEL_GROUP, opened_at) diff --git a/tests/e2e/quota_management/spend_tracking/cost_rows.py b/tests/e2e/quota_management/spend_tracking/cost_rows.py index 87af54fe83f..8b15c555c66 100644 --- a/tests/e2e/quota_management/spend_tracking/cost_rows.py +++ b/tests/e2e/quota_management/spend_tracking/cost_rows.py @@ -36,6 +36,7 @@ from pydantic import BaseModel, RootModel from e2e_config import unique_marker from e2e_http import Success +from e2e_metadata import step from lifecycle import ResourceManager from models import LiteLLMParamsBody, SpendLogsParams from proxy_client import ProxyClient @@ -133,6 +134,7 @@ def assert_fresh_tokens_billed_at(row: CostRow, input_rate: float) -> None: ) +@step("Wait for the request's cost breakdown in /spend/logs") def poll_cost_row(proxy: ProxyClient, request_id: str) -> CostRow | None: """Poll /spend/logs for the call's row until it lands with a cost breakdown (rows flush ~60s behind the call via proxy_batch_write_at); None on timeout.""" @@ -156,6 +158,7 @@ def poll_cost_row(proxy: ProxyClient, request_id: str) -> CostRow | None: return None +@step("Wait for a matching cost breakdown in the key's /spend/logs") def poll_cost_row_where( proxy: ProxyClient, api_key: str, predicate: Callable[[CostRow], bool] ) -> CostRow | None: @@ -182,6 +185,7 @@ def poll_cost_row_where( return None +@step("Add a deployment with custom rates that calls {litellm_params.model}") def register_priced_model( proxy: ProxyClient, resources: ResourceManager, diff --git a/tests/e2e/quota_management/spend_tracking/spend_e2e_client.py b/tests/e2e/quota_management/spend_tracking/spend_e2e_client.py index e607c12b731..60372a35afe 100644 --- a/tests/e2e/quota_management/spend_tracking/spend_e2e_client.py +++ b/tests/e2e/quota_management/spend_tracking/spend_e2e_client.py @@ -31,6 +31,7 @@ from e2e_http import ( is_ok, unwrap, ) +from e2e_metadata import step from models import ( AnthropicMessagesBody, ChatBody, @@ -264,6 +265,7 @@ def _chat_body( class SpendClient: proxy: ProxyClient + @step('Send a /chat/completions request to {model} with the prompt "{content}"') def chat( self, key: str, @@ -280,6 +282,7 @@ class SpendClient: _chat_body(model, content, max_tokens=max_tokens, tags=tags, user=user, cache=cache), ) + @step('Send a streaming /chat/completions request to {model} with the prompt "{content}"') def chat_stream( self, key: str, model: str, content: str, *, max_tokens: int | None = None ) -> StreamingResponse: @@ -287,6 +290,7 @@ class SpendClient: key, _chat_body(model, content, max_tokens=max_tokens, stream=True) ) + @step('Send a streaming /v1/messages request to {model} with the prompt "{content}"') def messages_stream( self, key: str, model: str, content: str, *, max_tokens: int ) -> StreamingResponse: @@ -300,9 +304,11 @@ class SpendClient: ), ) + @step('Send an /embeddings request to {model} for "{content}"') def embed(self, key: str, model: str, content: str) -> Result[EmbedResponse]: return self.proxy.embed(key, EmbedBody(model=model, input=content)) + @step("Wait for at least {min_rows} of the key's spend logs in /spend/logs") def poll_logs_for_key( self, key: str, @@ -314,6 +320,7 @@ class SpendClient: key, min_rows=min_rows, predicate=predicate ) + @step('Estimate the cost of sending "{content}" to {model} with /spend/calculate') def calculate_spend(self, model: str, content: str) -> float: return unwrap( self.proxy.transport.post( @@ -326,6 +333,7 @@ class SpendClient: ) ).cost + @step("Read the spend per tag from /spend/tags") def spend_by_tags(self) -> list[TagSpend]: result = self.proxy.transport.get( "/spend/tags", @@ -339,6 +347,7 @@ class SpendClient: case _: return [] + @step("Wait for the tag {tag} to reach the expected spend in /spend/tags") def poll_tag_spend(self, tag: str, *, minimum: float = 0.0) -> TagSpend | None: """Poll /spend/tags until the tag's aggregate reaches `minimum`; last seen.""" deadline = time.monotonic() + self.proxy.poll_timeout @@ -354,6 +363,7 @@ class SpendClient: time.sleep(self.proxy.poll_interval) return entry + @step("Wait for the key's spend in /key/info to reach the expected minimum") def poll_key_spend(self, key: str, *, minimum: float = 0.0) -> float: deadline = time.monotonic() + self.proxy.poll_timeout spend = 0.0 @@ -364,6 +374,7 @@ class SpendClient: time.sleep(self.proxy.poll_interval) return spend + @step("Read the team's spend from /team/info") def team_spend(self, team_id: str) -> float: return ( unwrap( @@ -377,6 +388,7 @@ class SpendClient: or 0.0 ) + @step("Wait for the team's spend in /team/info to reach the expected minimum") def poll_team_spend(self, team_id: str, *, minimum: float = 0.0) -> float: outcome: Final = await_converged( lambda: self.team_spend(team_id), @@ -388,6 +400,7 @@ class SpendClient: ) return outcome.result if isinstance(outcome, Converged) else outcome.last_result + @step("Read the end user's spend from /customer/info") def customer_spend(self, customer_id: str) -> float: """0.0 until the spend writer has upserted the end-user row, which /customer/info 404s before.""" looked_up: Final = self.proxy.transport.get( @@ -402,6 +415,7 @@ class SpendClient: case _: return 0.0 + @step("Wait for the end user's spend in /customer/info to go above {minimum}") def poll_customer_spend(self, customer_id: str, *, minimum: float = 0.0) -> float: outcome: Final = await_converged( lambda: self.customer_spend(customer_id), @@ -413,6 +427,7 @@ class SpendClient: ) return outcome.result if isinstance(outcome, Converged) else outcome.last_result + @step("Scrape /metrics/ on every proxy replica") def scrape_metrics(self) -> Mapping[str, ProbeResult]: """GET /metrics/ on every replica in PROXY_REPLICA_URLS, keyed by replica. The counter is per pod, so the union of the replicas is the fleet's exposition; the @@ -425,6 +440,7 @@ class SpendClient: } ) + @step("Read page {page} of /spend/logs/v2 at a page size of {page_size}") def spend_logs_page( self, *, api_key: str | None, page: int, page_size: int ) -> SpendLogsPage: @@ -447,9 +463,11 @@ class SpendClient: ) ) + @step("Call the management route {path}") def probe(self, path: str, *, params: DateRangeParams) -> ProbeResult: return self.proxy.transport.probe(path, params=params) + @step("Call the management route {path} until it answers successfully") def probe_until_healthy(self, path: str, *, params: DateRangeParams) -> ProbeResult: outcome: Final = await_converged( lambda: self.probe(path, params=params), @@ -461,6 +479,7 @@ class SpendClient: ) return outcome.result if isinstance(outcome, Converged) else outcome.last_result + @step("Create an internal user with the role {role}") def create_user(self, *, email: str, role: UserRole, user_id: str) -> str: return unwrap( self.proxy.transport.post( @@ -471,6 +490,7 @@ class SpendClient: ) ).user_id + @step("Delete the internal user") def delete_user(self, user_id: str) -> None: _ = unwrap( self.proxy.transport.post( @@ -481,6 +501,7 @@ class SpendClient: ) ) + @step("Generate a virtual key with {body}") def generate_key_record(self, body: KeyGenerateBody) -> KeyGenerateResponse: return unwrap( self.proxy.transport.post( @@ -491,6 +512,7 @@ class SpendClient: ) ) + @step('Send a /chat/completions request to {model} with the prompt "{content}"') def send_chat(self, key: str, model: str, content: str, *, max_tokens: int) -> StreamingResponse: return self.proxy.transport.send( "/chat/completions", @@ -498,6 +520,7 @@ class SpendClient: json=_chat_body(model, content, max_tokens=max_tokens), ) + @step('Send a /queue/chat/completions request to {model} with the prompt "{content}"') def send_queued_chat(self, key: str, model: str, content: str, *, max_tokens: int) -> StreamingResponse: return self.proxy.transport.send( "/queue/chat/completions", @@ -509,6 +532,7 @@ class SpendClient: ), ) + @step('Send a /v1/messages request to {model} with the prompt "{content}"') def send_messages(self, key: str, model: str, content: str, *, max_tokens: int) -> StreamingResponse: return self.proxy.transport.send( "/v1/messages", @@ -520,9 +544,11 @@ class SpendClient: ), ) + @step('Send a /v1/responses request to {model} with the prompt "{content}"') def send_responses(self, key: str, model: str, content: str) -> StreamingResponse: return self.send_responses_with_headers(self.proxy.transport.bearer(key), model, content) + @step('Send a /v1/responses request to {model} with custom headers and the prompt "{content}"') def send_responses_with_headers(self, headers: AuthHeaders, model: str, content: str) -> StreamingResponse: return self.proxy.transport.send( "/v1/responses", @@ -530,6 +556,7 @@ class SpendClient: json=ResponsesBody(model=model, input=content), ) + @step('Send an /embeddings request to {model} for "{content}"') def send_embed(self, key: str, model: str, content: str) -> StreamingResponse: return self.proxy.transport.send( "/embeddings", @@ -537,6 +564,7 @@ class SpendClient: json=EmbedBody(model=model, input=content), ) + @step('Send a Gemini generateContent request to {model} through /gemini with the prompt "{content}"') def send_gemini_generate(self, key: str, model: str, content: str, *, max_tokens: int) -> StreamingResponse: return self.proxy.transport.send( f"/gemini/v1beta/models/{model}:generateContent", @@ -547,6 +575,7 @@ class SpendClient: ), ) + @step("Upload a batch input file for {model} to /v1/files") def upload_batch_file(self, key: str, model: str, content: bytes) -> FileObject: return unwrap( self.proxy.transport.upload( @@ -560,6 +589,7 @@ class SpendClient: ) ) + @step("Create a batch for {body.model} on /v1/batches") def create_batch(self, key: str, body: BatchCreateBody) -> BatchObject: return unwrap( self.proxy.transport.post( @@ -570,6 +600,7 @@ class SpendClient: ) ) + @step("Retrieve the {provider} batch from /v1/batches") def retrieve_batch(self, key: str, batch_id: str, *, provider: str) -> BatchObject: return unwrap( self.proxy.transport.get( @@ -580,6 +611,7 @@ class SpendClient: ) ) + @step("Post a callback log for {payload.model} to /v1/rust_control_plane/logs") def replay_callback_log(self, key: str, payload: CallbackLogPayload) -> CallbackLogsResponse: return unwrap( self.proxy.transport.post( @@ -590,12 +622,15 @@ class SpendClient: ) ) + @step("Run a health check on {model} with /health") def health(self, model: str) -> ProbeResult: return self.proxy.transport.probe("/health", params=HealthParams(model=model)) + @step("Read the key's daily activity from /user/daily/activity") def daily_activity_for_key(self, token: str, *, start: datetime, end: datetime) -> DailyActivityKeyBreakdown | None: return self._key_breakdown("/user/daily/activity", token, start=start, end=end) + @step("Read the key's usage export row from /user/daily/activity/aggregated") def usage_export_row_for_key( self, token: str, *, start: datetime, end: datetime ) -> DailyActivityKeyBreakdown | None: @@ -623,11 +658,13 @@ class SpendClient: None, ) + @step("Wait for at least {min_requests} of the key's requests in /user/daily/activity") def poll_daily_activity_for_key( self, token: str, *, start: datetime, end: datetime, min_requests: int ) -> DailyActivityKeyBreakdown | None: return self._poll_key_breakdown(lambda: self.daily_activity_for_key(token, start=start, end=end), min_requests) + @step("Wait for at least {min_requests} of the key's requests in /user/daily/activity/aggregated") def poll_usage_export_row_for_key( self, token: str, *, start: datetime, end: datetime, min_requests: int ) -> DailyActivityKeyBreakdown | None: @@ -648,6 +685,7 @@ class SpendClient: ) return outcome.result if isinstance(outcome, Converged) else outcome.last_result + @step("Read the OpenAPI schema from /openapi.json") def openapi(self) -> OpenAPISchema: return unwrap( self.proxy.transport.get( diff --git a/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py b/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py index f313325dbda..96f1fbd4a7a 100644 --- a/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py +++ b/tests/e2e/quota_management/spend_tracking/spend_reconciliation.py @@ -5,6 +5,7 @@ from typing import Final from e2e_config import provider_edge_base, unique_marker from e2e_http import unwrap +from e2e_metadata import step from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody, LiteLLMParamsBody, TeamNewBody from spend_e2e_client import SpendClient @@ -33,6 +34,10 @@ class TeamTraffic: return self.prompt_tokens * INPUT_RATE + self.completion_tokens * OUTPUT_RATE +@step( + "Add a priced deployment, then create two teams with one key each" + " and send 7 /chat/completions requests per key, 6 of them at once" +) def create_traffic(client: SpendClient, resources: ResourceManager) -> tuple[TeamTraffic, ...]: base: Final = provider_edge_base("openai") model: Final = f"e2e-reconciliation-{unique_marker()}" @@ -85,6 +90,7 @@ def create_traffic(client: SpendClient, resources: ResourceManager) -> tuple[Tea return tuple(team_traffic() for _ in range(2)) +@step("Check that the /spend/logs rows of team {traffic.team_id} match each response's tokens and cost") def assert_logs_match(client: SpendClient, traffic: TeamTraffic) -> None: expected_ids: Final = frozenset(response.id for response in traffic.responses) assert len(expected_ids) == len(traffic.responses), "responses must have distinct IDs" diff --git a/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py b/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py index 76d80b1aab8..31c9b90d209 100644 --- a/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_service_tier_pricing_e2e.py @@ -36,7 +36,7 @@ from cost_rows import ( ) from e2e_config import CHEAP_OPENAI_MODEL, unique_marker from e2e_http import unwrap -from e2e_metadata import Capability, Domain, Mode, Provider, Subject, meta +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ( AnthropicMessagesBody, @@ -184,6 +184,15 @@ class TestServiceTierPricing: assert_total_is_sum_of_components(row) @pytest.mark.covers("quota_management.spend_tracking.service_tier_stream.records_served_tier") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(BACKEND,), + mode=Mode.STREAM, + ) + ) def test_streamed_call_records_and_bills_the_served_tier( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -232,6 +241,15 @@ class TestServiceTierPricing: assert_total_is_sum_of_components(row) @pytest.mark.covers("llm.chat_completions.openai.service_tier.stream.echoes_served_tier") + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.CHAT_COMPLETIONS, + providers=(Provider.OPENAI,), + models=(STREAM_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_every_streamed_chunk_carries_the_served_tier( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -262,6 +280,15 @@ class TestServiceTierPricing: ) @pytest.mark.covers("quota_management.spend_tracking.service_tier_stream.responses_records_served_tier") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(STREAM_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_responses_stream_records_the_served_tier( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -308,6 +335,15 @@ class TestServiceTierPricing: assert_fresh_tokens_billed_at(row, INPUT_RATE_FOR_PRICING_BASIS[pricing_basis]) @pytest.mark.covers("quota_management.spend_tracking.service_tier_stream.messages_records_served_tier") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.MESSAGES, + providers=(Provider.OPENAI,), + models=(STREAM_BACKEND,), + mode=Mode.STREAM, + ) + ) def test_messages_stream_records_the_served_tier( self, client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: diff --git a/tests/e2e/quota_management/spend_tracking/test_spend_routes.py b/tests/e2e/quota_management/spend_tracking/test_spend_routes.py index 7b5db9ccd27..1b2069abdc6 100644 --- a/tests/e2e/quota_management/spend_tracking/test_spend_routes.py +++ b/tests/e2e/quota_management/spend_tracking/test_spend_routes.py @@ -163,6 +163,12 @@ def test_schema_listed_spend_routes_are_responsive(client: SpendClient) -> None: assert not offenders, "non-responsive schema spend routes:\n" + "\n".join(offenders) +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.SPEND_REPORTING, + ) +) def test_capture_rate_reports_or_names_the_missing_billing_key(client: SpendClient) -> None: result: Final = client.probe(_CAPTURE_RATE_ROUTE, params=_date_range()) print(f"{_CAPTURE_RATE_ROUTE} -> {result.status_code}\n{result.body[:600]}") diff --git a/tests/e2e/quota_management/spend_tracking/test_spend_surface_consistency_e2e.py b/tests/e2e/quota_management/spend_tracking/test_spend_surface_consistency_e2e.py index 9033b9d75c4..63ba785a2ee 100644 --- a/tests/e2e/quota_management/spend_tracking/test_spend_surface_consistency_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_spend_surface_consistency_e2e.py @@ -32,6 +32,7 @@ from typing import Final import pytest from e2e_config import provider_edge_base, unique_marker from e2e_http import ProbeResult +from e2e_metadata import Domain, Mode, Provider, Subject, meta from lifecycle import ResourceManager from models import ChatBody, ChatMessage, KeyGenerateBody, LiteLLMParamsBody, TeamNewBody from prometheus_client.parser import text_string_to_metric_families @@ -41,6 +42,7 @@ from spend_reconciliation import INPUT_RATE, OUTPUT_RATE pytestmark = pytest.mark.e2e +BACKEND: Final = "openai/gpt-5.6-luna" SPEND_METRIC: Final = "litellm_spend_metric_total" KEY_HASH_LABEL: Final = "hashed_api_key" TEAM_LABEL: Final = "team" @@ -86,6 +88,14 @@ def _same_spend(actual: float | None, expected: float) -> bool: class TestSpendSurfaceConsistency: @pytest.mark.replayable @pytest.mark.covers("quota_management.spend_tracking.surface_consistency.matches_every_surface") + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(BACKEND,), + mode=Mode.NONSTREAM, + ) + ) def test_one_request_lands_the_same_spend_on_every_surface( self, client: SpendClient, resources: ResourceManager ) -> None: @@ -96,7 +106,7 @@ class TestSpendSurfaceConsistency: model_id: Final = client.proxy.create_model( model, LiteLLMParamsBody( - model="openai/gpt-5.6-luna", + model=BACKEND, api_key="os.environ/OPENAI_API_KEY", api_base=None if base is None else f"{base}/v1", input_cost_per_token=INPUT_RATE, diff --git a/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py b/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py index a4c37c2df94..c7ef81f826a 100644 --- a/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_spend_tracking_e2e.py @@ -42,6 +42,7 @@ CLAUDE_MODEL = "claude-haiku-4-5" CODEX_MODEL = "openai-responses-codex" EMBEDDING_MODEL = "openai-text-embedding-3-small" OPENAI_BACKEND = "openai/gpt-5.5" +ANTHROPIC_BACKEND: Final = "anthropic/claude-haiku-4-5" def _approx_equal(actual: float, expected: float) -> bool: @@ -534,6 +535,15 @@ def test_end_user_spend_attributed_on_row( @pytest.mark.covers("quota_management.spend_tracking.end_user.attributes_responses_header") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.RESPONSES, + providers=(Provider.OPENAI,), + models=(CODEX_MODEL,), + mode=Mode.NONSTREAM, + ) +) @pytest.mark.parametrize("header", ["x-litellm-customer-id", "x-litellm-end-user-id"]) def test_end_user_header_attributes_responses_row( client: SpendClient, scoped_key: str, resources: ResourceManager, header: str @@ -659,6 +669,14 @@ def test_failure_call_writes_failure_status_row( @pytest.mark.covers("quota_management.spend_tracking.failure.writes_normalized_error") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.ANTHROPIC, Provider.OPENAI), + models=(OPENAI_BACKEND, ANTHROPIC_BACKEND), + mode=Mode.NONSTREAM, + ) +) def test_failure_rows_share_normalized_error_across_provider_wording( client: SpendClient, resources: ResourceManager, scoped_key: str ) -> None: @@ -668,7 +686,7 @@ def test_failure_rows_share_normalized_error_across_provider_wording( marker = unique_marker() deployments: Final = ( (f"e2e-norm-openai-{marker}", OPENAI_BACKEND), - (f"e2e-norm-anthropic-{marker}", "anthropic/claude-haiku-4-5"), + (f"e2e-norm-anthropic-{marker}", ANTHROPIC_BACKEND), ) for name, provider_model in deployments: model_id = client.proxy.create_model( @@ -700,6 +718,14 @@ def test_failure_rows_share_normalized_error_across_provider_wording( @pytest.mark.covers("quota_management.spend_tracking.failure.attributes_provider") +@meta( + Subject( + domain=Domain.SPEND_BUDGETS, + providers=(Provider.OPENAI,), + models=(OPENAI_BACKEND,), + mode=Mode.NONSTREAM, + ) +) def test_pre_call_rejection_row_attributes_provider_and_model_id( client: SpendClient, resources: ResourceManager ) -> None: diff --git a/tests/e2e/quota_management/spend_tracking/test_websearch_interception_session_e2e.py b/tests/e2e/quota_management/spend_tracking/test_websearch_interception_session_e2e.py index 6352ab67c3c..ab7c2370365 100644 --- a/tests/e2e/quota_management/spend_tracking/test_websearch_interception_session_e2e.py +++ b/tests/e2e/quota_management/spend_tracking/test_websearch_interception_session_e2e.py @@ -17,6 +17,7 @@ from typing import Final, Literal import pytest from e2e_config import unique_marker from e2e_http import unwrap +from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager from models import ( AnthropicContentBlock, @@ -56,6 +57,16 @@ class TestWebSearchInterceptionSession: "quota_management.spend_tracking.websearch_interception.bills_under_request_session", exercised_on=("messages",), ) + @meta( + Subject( + domain=Domain.SPEND_BUDGETS, + route=Route.MESSAGES, + providers=(Provider.BEDROCK, Provider.PERPLEXITY), + models=(BEDROCK_INVOKE_BACKEND,), + capabilities=(Capability.WEB_SEARCH,), + mode=Mode.NONSTREAM, + ) + ) def test_intercepted_search_is_billed_under_the_request_session( self, proxy: ProxyClient, resources: ResourceManager ) -> None: From 0338498067c60720cac27f5df176def143173bff Mon Sep 17 00:00:00 2001 From: tin-berri Date: Wed, 7 Oct 2026 11:00:31 -0700 Subject: [PATCH 09/13] fix(router): preserve native baseline identity and accounting (#44960) * fix(router): preserve native baseline identity and accounting * fix(router): preserve injected system caches in native baselines * fix(router): prepare native baselines through shared request owners * fix(router): capture native baseline fields from the provider schema * refactor(router): reuse native provider parameter discovery * fix(router): keep long native baselines and abstain after compaction Message history no longer counts against the settings snapshot budget, so long and non-ASCII native sessions keep modeled baselines. Selected-tier compaction now abstains because the baseline would otherwise inherit the compacted history. Co-Authored-By: Claude Opus 5.5 * fix(router): judge implicit caching against the selected request The implicit-cache guard compared the selected response's cache usage with the projected baseline's breakpoints, so selected-tier cache markers made a usable unmarked baseline plan look like unexplained caching. The guard now checks the selected wire request. Also removes a stamp-reuse branch that could never run because routing clears the stamp first; every pass already captures caller settings from fresh kwargs. Co-Authored-By: Claude Opus 5.5 --------- Co-authored-by: Claude Opus 5.5 --- litellm/integrations/custom_logger.py | 18 +- litellm/litellm_core_utils/logging_worker.py | 27 +- litellm/litellm_core_utils/redact_messages.py | 27 +- .../pass_through/messages/handler.py | 63 +- .../anthropic/pass_through/messages/utils.py | 52 +- .../llms/anthropic/prompt_cache_prediction.py | 64 +- litellm/proxy/db/baseline_accounting.py | 3 +- .../proxy/hooks/autorouter_baseline_cache.py | 229 +++-- .../spend_tracking/baseline_accounting.py | 18 +- litellm/router.py | 6 +- .../complexity_router/context_compaction.py | 5 + litellm/router_utils/baseline_request.py | 153 ++++ litellm/types/router.py | 5 +- .../code_coverage_tests/recursive_detector.py | 1 + .../spend/test_baseline_accounting.py | 7 +- .../litellm_core_utils/test_logging_worker.py | 60 +- .../test_redact_messages.py | 176 ++-- ...erimental_pass_through_messages_handler.py | 18 +- .../test_request_optional_param_utils.py | 10 +- .../hooks/test_autorouter_baseline_cache.py | 804 +++++++++++++++++- .../test_baseline_accounting.py | 80 +- .../router_utils/test_baseline_request.py | 59 ++ 22 files changed, 1608 insertions(+), 277 deletions(-) create mode 100644 litellm/router_utils/baseline_request.py create mode 100644 tests/unit/router_utils/test_baseline_request.py diff --git a/litellm/integrations/custom_logger.py b/litellm/integrations/custom_logger.py index 6a4987298c4..45fce665bf0 100644 --- a/litellm/integrations/custom_logger.py +++ b/litellm/integrations/custom_logger.py @@ -909,7 +909,8 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac """ import litellm from litellm import Choices, Message, ModelResponse - from litellm.litellm_core_utils.classifier_logging import CLASSIFIER_AUDIT_FIELDS, without_classifier_audit + from litellm.litellm_core_utils.classifier_logging import CLASSIFIER_AUDIT_FIELDS + from litellm.litellm_core_utils.redact_messages import redacted_litellm_params turn_off_message_logging: Final[bool] = getattr(self, "turn_off_message_logging", False) excluded_fields: Final[list[str] | None] = getattr(litellm, "standard_logging_payload_excluded_fields", None) @@ -918,9 +919,15 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac if turn_off_message_logging is False and not excluded_fields: return model_call_details + params: Final = model_call_details.get("litellm_params") + redacted_params: Final = ( + MappingProxyType({"litellm_params": redacted_litellm_params(params)}) + if turn_off_message_logging and isinstance(params, Mapping) + else EMPTY_MAPPING + ) standard_logging_object: Final = model_call_details.get("standard_logging_object") if standard_logging_object is None: - return model_call_details.copy() + return {**model_call_details, **redacted_params} # Make a copy of just the standard_logging_object to avoid modifying the original standard_logging_object_copy: Final = { @@ -960,13 +967,6 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac model_response_dict: Final = model_response.model_dump() standard_logging_object_copy["response"] = model_response_dict - params: Final = model_call_details.get("litellm_params") - request: Final = params.get("proxy_server_request") if isinstance(params, dict) else None - redacted_params: Final = ( - MappingProxyType({"litellm_params": {**params, "proxy_server_request": without_classifier_audit(request)}}) - if turn_off_message_logging and isinstance(params, dict) and isinstance(request, dict) - else EMPTY_MAPPING - ) return { **model_call_details, **redacted_params, diff --git a/litellm/litellm_core_utils/logging_worker.py b/litellm/litellm_core_utils/logging_worker.py index 03420b84c22..57da3f8dabe 100644 --- a/litellm/litellm_core_utils/logging_worker.py +++ b/litellm/litellm_core_utils/logging_worker.py @@ -23,6 +23,19 @@ from litellm.constants import ( MAX_TIME_TO_CLEAR_QUEUE, ) +_CALLBACK_DEADLINE: Final[contextvars.ContextVar[float | None]] = contextvars.ContextVar( + "logging_callback_deadline", default=None +) + + +def optional_callback_budget(maximum: float, *, fraction: float = 0.25) -> float: + deadline: Final = _CALLBACK_DEADLINE.get() + return ( + maximum + if deadline is None + else max(0.0, min(maximum, (deadline - asyncio.get_running_loop().time()) * fraction)) + ) + def _coroutine_name(coroutine: Coroutine) -> str: return getattr(coroutine, "__qualname__", None) or getattr(coroutine, "__name__", None) or type(coroutine).__name__ @@ -100,12 +113,20 @@ class LoggingWorker: return len(revived) def _run_coroutine_silently(self, loop: asyncio.AbstractEventLoop, coroutine: Coroutine) -> bool: + token: Final = _CALLBACK_DEADLINE.set(loop.time() + self.timeout) try: loop.run_until_complete(asyncio.wait_for(coroutine, timeout=self.timeout)) except (Exception, asyncio.CancelledError): # noqa: BLE001 # atexit flush must never break the user's program return False + finally: + _CALLBACK_DEADLINE.reset(token) return True + def _create_callback_task(self, task: LoggingTask) -> asyncio.Task[object]: + context: Final = task["context"].copy() + context.run(_CALLBACK_DEADLINE.set, asyncio.get_running_loop().time() + self.timeout) + return context.run(asyncio.create_task, task["coroutine"]) + @staticmethod def _drain_pending(queue: "asyncio.Queue[LoggingTask]") -> tuple[LoggingTask, ...]: """Pop every task still queued, without awaiting them, so they can be moved to another queue.""" @@ -172,7 +193,7 @@ class LoggingWorker: try: if self._queue is not None: # Run the coroutine in its original context - callback_task: Final = task["context"].run(asyncio.create_task, task["coroutine"]) + callback_task: Final = self._create_callback_task(task) try: await asyncio.wait_for(callback_task, timeout=self.timeout) except asyncio.TimeoutError as e: @@ -424,7 +445,7 @@ class LoggingWorker: try: await asyncio.wait_for( - task["context"].run(asyncio.create_task, task["coroutine"]), + self._create_callback_task(task), timeout=self.timeout, ) except Exception: @@ -517,7 +538,7 @@ class LoggingWorker: # Await the coroutine to properly execute and avoid "never awaited" warnings try: await asyncio.wait_for( - task["context"].run(asyncio.create_task, task["coroutine"]), + self._create_callback_task(task), timeout=self.timeout, ) except Exception: diff --git a/litellm/litellm_core_utils/redact_messages.py b/litellm/litellm_core_utils/redact_messages.py index 85ed0a40687..7c6c39abb76 100644 --- a/litellm/litellm_core_utils/redact_messages.py +++ b/litellm/litellm_core_utils/redact_messages.py @@ -11,6 +11,7 @@ import asyncio import copy import inspect from collections.abc import Mapping +from dataclasses import replace from typing import TYPE_CHECKING, Any, Final import litellm @@ -26,6 +27,7 @@ from litellm.llms.vertex_ai.common_utils import ( redact_vertex_ai_metadata_from_logged_object, ) from litellm.secret_managers.main import str_to_bool +from litellm.types.router import BaselineRouteStamp from litellm.types.utils import StandardCallbackDynamicParams if TYPE_CHECKING: @@ -252,6 +254,26 @@ def _redact_model_response_dict_choices(choices, redacted_str: str): _redact_choice_content(choice) +def _redacted_baseline_metadata(metadata: Mapping[str, object]) -> Mapping[str, object]: + route: Final = metadata.get("_autorouter_baseline_route") + if not isinstance(route, BaselineRouteStamp): + return metadata + return {**metadata, "_autorouter_baseline_route": replace(route, request_parameters=None)} + + +def redacted_litellm_params(params: Mapping[str, object]) -> dict[str, object]: + request: Final = params.get("proxy_server_request") + return { + **params, + **{ + key: _redacted_baseline_metadata(value) + for key, value in params.items() + if key in ("metadata", "litellm_metadata") and isinstance(value, Mapping) + }, + **({"proxy_server_request": without_classifier_audit(request)} if isinstance(request, Mapping) else {}), + } + + def perform_redaction(model_call_details: dict, result, redact_streaming_responses: bool = True): """ Performs the actual redaction on the logging object and result. @@ -262,9 +284,8 @@ def perform_redaction(model_call_details: dict, result, redact_streaming_respons """ # Redact model_call_details params: Final = model_call_details.get("litellm_params") - request: Final = params.get("proxy_server_request") if isinstance(params, dict) else None - if isinstance(params, dict) and isinstance(request, Mapping): - model_call_details["litellm_params"] = {**params, "proxy_server_request": without_classifier_audit(request)} + if isinstance(params, Mapping): + model_call_details["litellm_params"] = redacted_litellm_params(params) model_call_details["messages"] = [{"role": "user", "content": REDACTED_BY_LITELLM}] model_call_details["prompt"] = "" model_call_details["input"] = "" diff --git a/litellm/llms/anthropic/pass_through/messages/handler.py b/litellm/llms/anthropic/pass_through/messages/handler.py index 6d13e38aa45..4e7a154be67 100644 --- a/litellm/llms/anthropic/pass_through/messages/handler.py +++ b/litellm/llms/anthropic/pass_through/messages/handler.py @@ -15,9 +15,6 @@ import litellm from litellm.litellm_core_utils.exception_mapping_utils import exception_type from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.anthropic.common_utils import ( - flatten_unencrypted_web_search_results_in_anthropic_messages, - sanitize_tool_use_ids_in_anthropic_messages, - strip_empty_content_blocks_from_anthropic_messages, strip_provider_specific_fields_from_anthropic_messages, ) from litellm.llms.base_llm.anthropic_messages.transformation import ( @@ -36,9 +33,8 @@ from litellm.utils import ProviderConfigManager, client from ..adapters.handler import LiteLLMMessagesToCompletionTransformationHandler from ..responses_adapters.handler import LiteLLMMessagesToResponsesAPIHandler -from ..utils import is_reasoning_auto_summary_enabled from .interceptors import get_messages_interceptors -from .utils import AnthropicMessagesRequestUtils, mock_response +from .utils import AnthropicMessagesRequestUtils, mock_response, prepare_native_messages __all__ = ("anthropic_messages", "anthropic_messages_handler") @@ -251,28 +247,7 @@ async def anthropic_messages( Runs the empty-content-block sanitizer before any backend dispatch. """ - # Anthropic's API rejects requests containing empty / whitespace-only - # text content blocks ("messages: text content blocks must be - # non-empty") and empty thinking blocks ("each thinking block must - # contain thinking"). Multi-turn tool-use clients (e.g. Claude Code) - # routinely loop assistant responses that contain such blocks — an empty - # text block alongside tool_use, or an empty thinking block from a turn - # a non-Anthropic reasoning model served through the bridge — back as - # conversation history, which then causes the next /v1/messages call to - # 400. /v1/chat/completions already handles this in - # anthropic_messages_pt; sanitize the native Anthropic Messages path - # here for the same guarantee. See #22930. - messages = strip_empty_content_blocks_from_anthropic_messages(messages) - # Replay of cross-provider tool history (e.g. kimi -> Anthropic) may carry - # ids like ``functions.Bash:0`` that violate Anthropic's id pattern. - messages = sanitize_tool_use_ids_in_anthropic_messages(messages) - messages = flatten_unencrypted_web_search_results_in_anthropic_messages(messages) - - from litellm.integrations.anthropic_cache_control_hook import ( - AnthropicCacheControlHook, - ) - - messages, system = AnthropicCacheControlHook.maybe_inject_cache_control( + messages, system = prepare_native_messages( messages, system, kwargs, model=model, custom_llm_provider=custom_llm_provider, tools=tools, api_base=api_base ) @@ -454,23 +429,15 @@ def anthropic_messages_handler( """ from litellm.types.utils import LlmProviders - # Sanitize empty text blocks so the sync entry point - # (litellm.messages.create -> anthropic_messages_handler) gets the same - # protection as the async wrapper. The async wrapper already sanitized and - # does not reassign messages before dispatch, so it sets - # ``_litellm_messages_presanitized`` to skip this redundant second - # full-messages scan. Pop it so it never leaks into provider params. - if not kwargs.pop("_litellm_messages_presanitized", False): - messages = strip_empty_content_blocks_from_anthropic_messages(messages) - messages = sanitize_tool_use_ids_in_anthropic_messages(messages) - messages = flatten_unencrypted_web_search_results_in_anthropic_messages(messages) - - from litellm.integrations.anthropic_cache_control_hook import ( - AnthropicCacheControlHook, - ) - - messages, system = AnthropicCacheControlHook.maybe_inject_cache_control( - messages, system, kwargs, model=model, custom_llm_provider=custom_llm_provider, tools=tools, api_base=api_base + messages, system = prepare_native_messages( + messages, + system, + kwargs, + model=model, + custom_llm_provider=custom_llm_provider, + tools=tools, + api_base=api_base, + presanitized=bool(kwargs.pop("_litellm_messages_presanitized", False)), ) metadata = validate_anthropic_api_metadata(metadata) @@ -645,14 +612,6 @@ def anthropic_messages_handler( custom_llm_provider=custom_llm_provider, ) ) - if is_reasoning_auto_summary_enabled(): - thinking_param: Final = anthropic_messages_optional_request_params.get("thinking") - if isinstance(thinking_param, dict) and thinking_param.get("type") != "disabled": - anthropic_messages_optional_request_params["thinking"] = { - **thinking_param, - "display": "summarized", - } - resolved_api_base: Final = ( dynamic_api_base if dynamic_api_base is not None and anthropic_messages_provider_config.uses_get_llm_provider_api_base() diff --git a/litellm/llms/anthropic/pass_through/messages/utils.py b/litellm/llms/anthropic/pass_through/messages/utils.py index dfab0af8eaa..8371615baea 100644 --- a/litellm/llms/anthropic/pass_through/messages/utils.py +++ b/litellm/llms/anthropic/pass_through/messages/utils.py @@ -2,6 +2,15 @@ from collections.abc import Iterable, Mapping, Sequence from functools import lru_cache from typing import TYPE_CHECKING, Any, Final, cast, get_type_hints +from pydantic import JsonValue + +from litellm.integrations.anthropic_cache_control_hook import AnthropicCacheControlHook +from litellm.llms.anthropic.common_utils import ( + flatten_unencrypted_web_search_results_in_anthropic_messages, + sanitize_tool_use_ids_in_anthropic_messages, + strip_empty_content_blocks_from_anthropic_messages, +) +from litellm.llms.anthropic.pass_through.utils import is_reasoning_auto_summary_enabled from litellm.types.llms.anthropic import ( AnthropicMessagesRequestOptionalParams, AnthropicStopDetails, @@ -119,8 +128,40 @@ def anthropic_system_to_openai_message(system: object) -> ChatCompletionSystemMe return ChatCompletionSystemMessage(role="system", content=system) +def prepare_native_messages( + messages: list[dict[str, JsonValue]], + system: str | list[dict[str, JsonValue]] | None, + kwargs: dict[str, object], + *, + model: str, + custom_llm_provider: str | None = None, + tools: list[dict[str, JsonValue]] | None = None, + api_base: str | None = None, + presanitized: bool = False, +) -> tuple[list[dict[str, JsonValue]], str | list[dict[str, JsonValue]] | None]: + normalized: Final = ( + messages + if presanitized + else flatten_unencrypted_web_search_results_in_anthropic_messages( + sanitize_tool_use_ids_in_anthropic_messages(strip_empty_content_blocks_from_anthropic_messages(messages)) + ) + ) + return cast( # cast-ok: legacy normalizers and injection preserve the JSON message and system shapes + tuple[list[dict[str, JsonValue]], str | list[dict[str, JsonValue]] | None], + AnthropicCacheControlHook.maybe_inject_cache_control( + normalized, + system, + kwargs, + model=model, + custom_llm_provider=custom_llm_provider, + tools=tools, + api_base=api_base, + ), + ) + + @lru_cache(maxsize=1) -def _anthropic_messages_optional_param_keys() -> frozenset[str]: +def anthropic_messages_optional_param_keys() -> frozenset[str]: """ Valid AnthropicMessagesRequestOptionalParams keys. @@ -152,7 +193,7 @@ class AnthropicMessagesRequestUtils: Returns: AnthropicMessagesRequestOptionalParams instance with only the valid parameters """ - valid_keys: Final = _anthropic_messages_optional_param_keys() + valid_keys: Final = anthropic_messages_optional_param_keys() filtered_params: Final = {k: v for k, v in params.items() if k in valid_keys and v is not None} if model is not None: from litellm.llms.anthropic.chat.transformation import AnthropicConfig @@ -174,6 +215,13 @@ class AnthropicMessagesRequestUtils: drop_params=drop_params, output_key=param, ) + if is_reasoning_auto_summary_enabled(): + thinking_param: Final = filtered_params.get("thinking") + if isinstance(thinking_param, dict) and thinking_param.get("type") != "disabled": + return cast( + AnthropicMessagesRequestOptionalParams, + {**filtered_params, "thinking": {**thinking_param, "display": "summarized"}}, + ) return cast(AnthropicMessagesRequestOptionalParams, filtered_params) diff --git a/litellm/llms/anthropic/prompt_cache_prediction.py b/litellm/llms/anthropic/prompt_cache_prediction.py index 00fe56a5e39..634eb264b38 100644 --- a/litellm/llms/anthropic/prompt_cache_prediction.py +++ b/litellm/llms/anthropic/prompt_cache_prediction.py @@ -5,6 +5,7 @@ import hashlib import json from collections.abc import Mapping, Sequence from dataclasses import dataclass, field +from functools import reduce from itertools import accumulate, groupby from types import MappingProxyType from typing import Annotated, Final, Literal, Protocol, TypeAlias @@ -13,20 +14,32 @@ import httpx from pydantic import ConfigDict, Field, JsonValue, StrictInt, TypeAdapter, ValidationError import litellm -from litellm.llms.anthropic.common_utils import AnthropicModelInfo, is_anthropic_oauth_key +from litellm.litellm_core_utils.dot_notation_indexing import delete_nested_value +from litellm.llms.anthropic.common_utils import ( + AnthropicModelInfo, + is_anthropic_oauth_key, + strip_provider_specific_fields_from_anthropic_messages, +) from litellm.llms.anthropic.count_tokens.handler import AnthropicCountTokensHandler from litellm.llms.anthropic.count_tokens.transformation import COUNT_TOKEN_OPTION_NAMES from litellm.llms.anthropic.pass_through.messages.transformation import ( DEFAULT_ANTHROPIC_API_VERSION, AnthropicMessagesConfig, ) +from litellm.llms.anthropic.pass_through.messages.utils import AnthropicMessagesRequestUtils, prepare_native_messages +from litellm.router_utils.baseline_request import ( + BASELINE_PARAMETERS, + capture_baseline_parameters, +) from litellm.types.llms.base import LiteLLMBaseModel -from litellm.types.router import LiteLLM_Params +from litellm.types.router import GenericLiteLLMParams, LiteLLM_Params from litellm.types.utils import ModelResponse from litellm.utils import supports_thinking_cache_preservation _JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) _HEADERS: Final = TypeAdapter(dict[str, str]) +_MESSAGES: Final = TypeAdapter(list[dict[str, JsonValue]]) +_SYSTEM: Final = TypeAdapter(str | list[dict[str, JsonValue]] | None) _counter: Final = AnthropicCountTokensHandler() @@ -325,7 +338,7 @@ def _entry_fingerprint(fingerprint: str, ttl_seconds: int) -> str: def parse_cache_plan(body: Mapping[str, JsonValue]) -> PromptCachePlan | UnsupportedCachePlan: try: - request: Final = _PlanRequest.model_validate(body) + request: Final = _PlanRequest.model_validate(dict(body)) positions: Final = _positions(body) except ValidationError: return UnsupportedCachePlan("unsupported_prompt_shape") @@ -618,13 +631,56 @@ def resolve_baseline_prediction_target(params: LiteLLM_Params) -> NativePredicti return _resolve_prediction_target(params, allow_configured_endpoint=True) +def prepare_native_baseline_body(request: Mapping[str, object], model: str) -> Mapping[str, JsonValue] | None: + parameters: Final = capture_baseline_parameters(request) + if parameters is None: + return None + source: Final = {**parameters, "messages": request.get("messages"), "stream": request.get("stream", False)} + try: + owned: Final = _JSON_OBJECT.validate_python(source) + context: Final = {**{k: v for k, v in request.items() if k not in ("metadata", "litellm_metadata")}, **owned} + resolved_model: Final = litellm.get_llm_provider(model=model, custom_llm_provider="anthropic")[0] + messages, system = prepare_native_messages( + _MESSAGES.validate_python(owned.get("messages")), + _SYSTEM.validate_python(owned.get("system")), + context, + model=resolved_model, + custom_llm_provider="anthropic", + tools=_MESSAGES.validate_python(owned.get("tools") or []), + ) + options: Final = AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param( + {**owned, "system": system}, + model=resolved_model, + custom_llm_provider="anthropic", + drop_params=owned.get("drop_params") is True, + ) + filtered: Final = reduce( + delete_nested_value, + TypeAdapter(tuple[str, ...]).validate_python(owned.get("additional_drop_params") or ()), + dict(options), + ) + body: Final = AnthropicMessagesConfig().transform_anthropic_messages_request( + model=resolved_model, + messages=strip_provider_specific_fields_from_anthropic_messages(messages), + anthropic_messages_optional_request_params=filtered, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + return MappingProxyType(_JSON_OBJECT.validate_python(body)) + except Exception: # noqa: BLE001 # an unsupported hypothetical request is unavailable, never an inference failure + return None + + def _resolve_prediction_target( params: LiteLLM_Params, *, allow_configured_endpoint: bool, ) -> NativePredictionTarget | UnsupportedPredictionTarget: configured_options: Final = frozenset(params.model_dump(exclude_defaults=True, exclude_none=True)) - if configured_options - _DEPLOYMENT_OPTIONS: + allowed: Final = ( + _DEPLOYMENT_OPTIONS | frozenset(BASELINE_PARAMETERS) if allow_configured_endpoint else _DEPLOYMENT_OPTIONS + ) + if configured_options - allowed: return UnsupportedPredictionTarget("unsupported_deployment_configuration") api_base: Final = AnthropicModelInfo.get_api_base(params.api_base) if not allow_configured_endpoint and api_base not in ( diff --git a/litellm/proxy/db/baseline_accounting.py b/litellm/proxy/db/baseline_accounting.py index b21b7c8a2b9..486717f9b93 100644 --- a/litellm/proxy/db/baseline_accounting.py +++ b/litellm/proxy/db/baseline_accounting.py @@ -223,7 +223,8 @@ ON CONFLICT (request_id) DO NOTHING _MARK_CONFLICT: Final = """ UPDATE "LiteLLM_AutoRouterBaselineObservation" SET conflicted = TRUE, revision = $4::bigint -WHERE request_id = $1 AND scope = $2 AND data <> $3 AND NOT conflicted +WHERE request_id = $1 AND scope = $2 AND NOT conflicted + AND (data::jsonb #- '{turn,turn_at}') <> ($3::jsonb #- '{turn,turn_at}') """ _READ_PAGE: Final = """ WITH times AS ( diff --git a/litellm/proxy/hooks/autorouter_baseline_cache.py b/litellm/proxy/hooks/autorouter_baseline_cache.py index e3cd6c67aa0..c4232609124 100644 --- a/litellm/proxy/hooks/autorouter_baseline_cache.py +++ b/litellm/proxy/hooks/autorouter_baseline_cache.py @@ -5,7 +5,7 @@ import hashlib import json import time from collections.abc import Callable, Mapping -from dataclasses import dataclass, replace +from dataclasses import dataclass, field, replace from datetime import datetime from types import MappingProxyType from typing import TYPE_CHECKING, Final @@ -16,9 +16,8 @@ from pydantic import ConfigDict, Field, JsonValue, TypeAdapter from litellm._logging import verbose_proxy_logger from litellm.constants import INTERNAL_CALL_ORIGIN_METADATA_KEY from litellm.integrations.custom_logger import CustomLogger -from litellm.litellm_core_utils.core_helpers import ( - get_litellm_metadata_from_kwargs, # pyright: ignore[reportUnknownVariableType] # legacy metadata boundary validated below -) +from litellm.litellm_core_utils.core_helpers import get_metadata_variable_name_from_kwargs +from litellm.litellm_core_utils.logging_worker import optional_callback_budget from litellm.llms.anthropic.prompt_cache_prediction import ( CountedPromptCachePlan, NativePredictionTarget, @@ -28,6 +27,7 @@ from litellm.llms.anthropic.prompt_cache_prediction import ( count_cache_plan, count_prompt_tokens, parse_cache_plan, + prepare_native_baseline_body, resolve_baseline_prediction_target, supported_baseline_recipient, supported_prediction_headers, @@ -37,6 +37,8 @@ from litellm.proxy.spend_tracking.savings import ( _effective_model_info, # pyright: ignore[reportPrivateUsage] # existing deployment-price owner _proxy_llm_router, # pyright: ignore[reportPrivateUsage] # existing optional proxy-router owner ) +from litellm.router_strategy.complexity_router.context_compaction import compaction_applied +from litellm.router_utils.baseline_request import baseline_request from litellm.types.llms.base import LiteLLMBaseModel from litellm.types.router import BaselineRouteStamp from litellm.types.utils import CallTypes, ModelInfo, Usage @@ -66,6 +68,9 @@ class CapturedBaselineObservation(LiteLLMBaseModel): prices: ModelInfo | None observation: BaselineObservation + def with_observation(self, observation: BaselineObservation) -> CapturedBaselineObservation: + return self.model_copy(update={"observation": observation}) + @dataclass(frozen=True, slots=True) class BaselineCacheContext: @@ -73,7 +78,10 @@ class BaselineCacheContext: capture: CapturedBaselineObservation target: NativePredictionTarget | UnsupportedPredictionTarget baseline_deployment_id: str + baseline_body: Mapping[str, JsonValue] | None = field(default=None, repr=False) + selected_body_digest: str | None = field(default=None, repr=False) invalidated: str | None = None + finalization: asyncio.Task[CapturedBaselineObservation] | None = field(default=None, repr=False, compare=False) class _Metadata(LiteLLMBaseModel): @@ -102,6 +110,10 @@ def _digest(value: object) -> str: return hashlib.sha256(json.dumps(value, sort_keys=True, separators=(",", ":")).encode()).hexdigest() +def _native_body_digest(body: Mapping[str, JsonValue]) -> str: + return _digest({key: value for key, value in body.items() if key not in ("metadata", "stream")}) + + class AutoRouterBaselineCache(CustomLogger): def __init__( self, @@ -124,12 +136,15 @@ class AutoRouterBaselineCache(CustomLogger): if not isinstance(logging_obj, Logging) or call_type != CallTypes.anthropic_messages: return try: - metadata: Final = _METADATA.validate_python(get_litellm_metadata_from_kwargs({"litellm_params": kwargs})) + raw_metadata: Final = kwargs.get(get_metadata_variable_name_from_kwargs(kwargs)) + metadata: Final = _METADATA.validate_python(raw_metadata) if isinstance(raw_metadata, Mapping) else {} if metadata.get(INTERNAL_CALL_ORIGIN_METADATA_KEY): return if logging_obj.baseline_cache_context is not None: await invalidate_baseline_cache(logging_obj, "retried_request") return + if not isinstance(metadata.get("_autorouter_baseline_route"), BaselineRouteStamp): + return request: Final = _Metadata.model_validate(metadata) session: Final = kwargs.get("litellm_session_id") or request.session_id or logging_obj.litellm_session_id if not isinstance(session, str) or not session or len(session) > 256: @@ -142,13 +157,27 @@ class AutoRouterBaselineCache(CustomLogger): prices: Final = _PRICES.validate_python( _effective_model_info(router, request.route.baseline_deployment_id, request.route.baseline_model) ) + params: Final = ( + _METADATA.validate_python(deployment.litellm_params.model_dump(mode="json")) if deployment else {} + ) + projected: Final = ( + baseline_request( + kwargs, + request.route.request_parameters, + params, + include_extra_body=False, + ) + if request.route.request_parameters is not None + else None + ) scope: Final = "autorouter-baseline:v3:" + _digest( ( + "baseline_request_v4", request.user_api_key_hash, session, request.route.router_name, request.route.baseline_deployment_id, - deployment.litellm_params.model_dump(mode="json"), + params, prices, ) ) @@ -170,8 +199,21 @@ class AutoRouterBaselineCache(CustomLogger): reason="incomplete_response", ), ) + selected_model: Final = kwargs.get("model") + selected_body: Final = prepare_native_baseline_body( + kwargs, selected_model if isinstance(selected_model, str) else logging_obj.model + ) logging_obj.baseline_cache_context = BaselineCacheContext( - self, capture, target, request.route.baseline_deployment_id + self, + capture, + target, + request.route.baseline_deployment_id, + prepare_native_baseline_body(projected, target.model) + if projected is not None and isinstance(target, NativePredictionTarget) + else None, + _native_body_digest(selected_body) + if selected_body is not None and not compaction_applied(kwargs) + else None, ) except Exception: # noqa: BLE001 # optional observation cannot fail inference verbose_proxy_logger.warning("Auto-router baseline observation could not be initialized") @@ -197,14 +239,16 @@ class AutoRouterBaselineCache(CustomLogger): async def plan( self, target: NativePredictionTarget, wire: httpx.Request, body: Mapping[str, JsonValue], usage: Usage | None ) -> tuple[CountedPromptCachePlan | None, str | None]: + deadline: Final = asyncio.get_running_loop().time() + optional_callback_budget(_COUNT_TIMEOUT, fraction=0.75) if not supported_prediction_headers(wire.headers): return None, "unsupported_request_headers" plan: Final = parse_cache_plan(body) if isinstance(plan, UnsupportedCachePlan): return None, plan.reason details: Final = usage.prompt_tokens_details if usage is not None else None + selected: Final = parse_cache_plan(_JSON_BODY.validate_json(wire.content)) if ( - not plan.breakpoints + (isinstance(selected, UnsupportedCachePlan) or not selected.breakpoints) and details is not None and ((details.cached_tokens or 0) + (details.cache_creation_tokens or 0)) ): @@ -215,7 +259,8 @@ class AutoRouterBaselineCache(CustomLogger): try: counted: Final = await asyncio.wait_for( - count_cache_plan(target.model, target.api_key, plan, token_counter=count), timeout=_COUNT_TIMEOUT + count_cache_plan(target.model, target.api_key, plan, token_counter=count), + timeout=max(0.0, deadline - asyncio.get_running_loop().time()), ) return (None, counted.reason) if isinstance(counted, UnsupportedCachePlan) else (counted, None) except TimeoutError: @@ -229,111 +274,121 @@ async def invalidate_baseline_cache(logging_obj: Logging, reason: str, *, comple if context is not None: logging_obj.baseline_cache_context = replace(context, invalidated=reason) logging_obj.baseline_observation = context.capture.model_copy( - update=MappingProxyType( - { - "observation": context.capture.observation.model_copy( - update=MappingProxyType( - { - "available_at": max(context.capture.observation.started_at, context.collector.clock()), - "reason": reason, - } - ) - ), - } - ) + update={ + "observation": context.capture.observation.model_copy( + update={ + "available_at": max(context.capture.observation.started_at, context.collector.clock()), + "reason": reason, + } + ), + } ) async def finalize_baseline_cache(logging_obj: Logging, response_obj: object) -> None: context: Final = logging_obj.baseline_cache_context - if context is None: + if context is None or logging_obj.baseline_observation is not None: return + task: Final = context.finalization or asyncio.create_task(_capture(context, logging_obj, response_obj)) + active: Final = context if context.finalization is not None else replace(context, finalization=task) + if context.finalization is None: + task.add_done_callback(_consume_finalization) + logging_obj.baseline_cache_context = active try: - capture: Final = await _capture(context, logging_obj, response_obj) - if logging_obj.baseline_cache_context is context: - logging_obj.baseline_observation = capture # rebind-ok: attach only to the captured request owner - except Exception: # noqa: BLE001 # observation failures must preserve inference and billing + capture: Final = await asyncio.shield(task) + if logging_obj.baseline_cache_context is active: + logging_obj.baseline_observation = capture # rebind-ok: publish only for the current attempt + except Exception: # noqa: BLE001 # estimation must preserve inference and billing await invalidate_baseline_cache(logging_obj, "observation_unavailable") -async def _capture( +def _consume_finalization(task: asyncio.Task[CapturedBaselineObservation]) -> None: + if not task.cancelled(): + task.exception() + + +async def _capture_native( context: BaselineCacheContext, logging_obj: Logging, response_obj: object ) -> CapturedBaselineObservation: - original: Final = context.capture.observation - details: Final = _METADATA.validate_python(logging_obj.model_call_details) - if details.get("cache_hit") is True: - return context.capture.model_copy( - update=MappingProxyType( - { - "observation": original.model_copy( - update=MappingProxyType({"outcome": "response_cache", "reason": "response_cache_hit"}) - ) - } - ) - ) - event: Final = _WireEvent.model_validate(details) + capture: Final = context.capture + original: Final = capture.observation + event: Final = _WireEvent.model_validate(logging_obj.model_call_details) wire: Final = event.httpx_response.request usage: Final = _ResponseUsage.model_validate(response_obj).usage + available: Final = event.completion_start_time.timestamp() complete: Final = ( event.custom_llm_provider == "anthropic" and event.httpx_response.status_code == 200 and (not event.stream or event.prompt_cache_response_complete) ) - started: Final = original.started_at - available: Final = event.completion_start_time.timestamp() - if context.invalidated or not complete or not started <= available <= context.collector.clock(): - return context.capture.model_copy( - update=MappingProxyType( - { - "observation": original.model_copy( - update=MappingProxyType( - { - "available_at": max(started, context.collector.clock()), - "reason": context.invalidated or "incomplete_response", - } - ) - ) + if context.invalidated or not complete or not original.started_at <= available <= context.collector.clock(): + return capture.with_observation( + original.model_copy( + update={ + "available_at": max(original.started_at, context.collector.clock()), + "reason": context.invalidated or "incomplete_response", } ) ) target: Final = context.target if isinstance(target, UnsupportedPredictionTarget) or not supported_baseline_recipient(target, wire): - return context.capture.model_copy( - update=MappingProxyType( - { - "observation": original.model_copy( - update=MappingProxyType( - { - "available_at": available, - "reason": target.reason - if isinstance(target, UnsupportedPredictionTarget) - else "unsupported_baseline_recipient", - } - ) - ) + return capture.with_observation( + original.model_copy( + update={ + "available_at": available, + "reason": target.reason + if isinstance(target, UnsupportedPredictionTarget) + else "unsupported_baseline_recipient", } ) ) body: Final = _JSON_BODY.validate_json(wire.content) - same: Final = ( - logging_obj.get_router_model_id() == context.baseline_deployment_id and body.get("model") == target.model - ) - plan, reason = await context.collector.plan(target, wire, body, usage) - minimum: Final = get_prompt_cache_min_tokens(target.model) - return context.capture.model_copy( - update=MappingProxyType( - { - "observation": BaselineObservation( - request_id=original.request_id, - started_at=started, - available_at=available, - outcome="complete", - baseline_equivalent=same, - usage=usage, - plan=plan, - minimum_cache_tokens=minimum, - reason=reason, - ) - } + projected: Final = context.baseline_body + if projected is None or context.selected_body_digest != _native_body_digest(body): + return capture.with_observation( + original.model_copy( + update={ + "available_at": available, + "usage": usage, + "reason": "unsupported_baseline_settings" + if projected is None + else "unsupported_request_transformation", + } + ) + ) + same: Final = logging_obj.get_router_model_id() == context.baseline_deployment_id and _native_body_digest( + projected + ) == _native_body_digest(body) + plan, reason = await context.collector.plan(target, wire, projected, usage) + return capture.with_observation( + BaselineObservation( + request_id=original.request_id, + started_at=original.started_at, + available_at=available, + outcome="complete", + baseline_equivalent=same, + usage=usage.model_copy(update={key: projected.get(key) for key in ("speed", "inference_geo")}) + if usage is not None and not same + else usage, + plan=plan, + reason=reason, + minimum_cache_tokens=get_prompt_cache_min_tokens(target.model), ) ) + + +async def _capture( + context: BaselineCacheContext, logging_obj: Logging, response_obj: object +) -> CapturedBaselineObservation: + if _METADATA.validate_python(logging_obj.model_call_details).get("cache_hit") is True: + return context.capture.model_copy( + update={ + "observation": context.capture.observation.model_copy( + update={ + "outcome": "response_cache", + "reason": "response_cache_hit", + } + ), + } + ) + return await _capture_native(context, logging_obj, response_obj) diff --git a/litellm/proxy/spend_tracking/baseline_accounting.py b/litellm/proxy/spend_tracking/baseline_accounting.py index 46a38f71260..e344c0d7169 100644 --- a/litellm/proxy/spend_tracking/baseline_accounting.py +++ b/litellm/proxy/spend_tracking/baseline_accounting.py @@ -71,12 +71,12 @@ def _complete_usage(usage: Usage | None) -> bool: if usage is None or usage.prompt_tokens < 0 or usage.completion_tokens < 0: return False details: Final = usage.prompt_tokens_details - if details is None: + if details is None or not hasattr(details, "cache_creation_tokens"): return False values: Final = (details.text_tokens, details.cached_tokens, details.cache_creation_tokens) if any(value is None or value < 0 for value in values): return False - split: Final = details.cache_creation_token_details + split: Final = details.cache_creation_token_details if hasattr(details, "cache_creation_token_details") else None writes: Final = details.cache_creation_tokens or 0 return ( usage.total_tokens == usage.prompt_tokens + usage.completion_tokens @@ -133,10 +133,13 @@ def _matches(entry: CacheEntry, markers: tuple[CountedBreakpoint, ...], started: def _ambiguous(entry: CacheEntry, markers: tuple[CountedBreakpoint, ...], started: float) -> bool: - return entry.available_at <= started < entry.expires_at and any( - entry.content_fingerprint in marker.lookback_content_fingerprints - and (entry.uncertain or entry.ttl_seconds != marker.ttl_seconds) - for marker in markers + matching: Final = tuple( + marker for marker in markers if entry.content_fingerprint in marker.lookback_content_fingerprints + ) + return ( + entry.available_at <= started < entry.expires_at + and bool(matching) + and (entry.uncertain or all(entry.ttl_seconds != marker.ttl_seconds for marker in matching)) ) @@ -261,7 +264,7 @@ def _writes(history: BaselineHistory, observation: BaselineObservation) -> tuple observation.started_at + hit.ttl_seconds, ), ) - if hit is not None and all(marker.fingerprint != hit.fingerprint for marker in markers) + if hit is not None else () ) return ( @@ -277,6 +280,7 @@ def _writes(history: BaselineHistory, observation: BaselineObservation) -> tuple uncertain=bool(ambiguous), ) for marker in markers + if hit is None or marker.prefix_tokens > hit.tokens ), ) diff --git a/litellm/router.py b/litellm/router.py index 77cff6254ea..ab9d8884a1d 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -14582,17 +14582,21 @@ class Router: to the deployment that actually served the request. Every attempt therefore writes or clears, never just writes. """ + from litellm.router_utils.baseline_request import capture_baseline_parameters from litellm.types.router import BaselineRouteStamp phase_attributes(routing_decision_attributes(routing_decision)) baseline_model: Final = routing_decision.get("savings_baseline_model") if routing_decision else None baseline_id: Final = routing_decision.get("savings_baseline_deployment_id") if routing_decision else None router_name: Final = routing_decision.get("router_model_name") if routing_decision else None + caller_parameters: Final = ( + capture_baseline_parameters(request_kwargs) if router_name and baseline_model else None + ) Router._stamp_or_clear_metadata_key( request_kwargs=request_kwargs, key="_autorouter_baseline_route", value=( - BaselineRouteStamp(router_name, baseline_model, baseline_id) + BaselineRouteStamp(router_name, baseline_model, baseline_id, caller_parameters) if router_name and baseline_model and baseline_id else None ), diff --git a/litellm/router_strategy/complexity_router/context_compaction.py b/litellm/router_strategy/complexity_router/context_compaction.py index d82090f3b99..6a1a63ea4b6 100644 --- a/litellm/router_strategy/complexity_router/context_compaction.py +++ b/litellm/router_strategy/complexity_router/context_compaction.py @@ -164,6 +164,11 @@ def compaction_pending(kwargs: Mapping[str, object] | None) -> bool: return isinstance(state, CompactionState) and state.config is not None and not _client_managed(kwargs or _EMPTY) +def compaction_applied(kwargs: Mapping[str, object]) -> bool: + state: Final = kwargs.get(_STATE_KEY) + return isinstance(state, CompactionState) and state.summary is not None + + def _reject(model: str, reason: str) -> NoReturn: from litellm.exceptions import BadRequestError diff --git a/litellm/router_utils/baseline_request.py b/litellm/router_utils/baseline_request.py new file mode 100644 index 00000000000..2f6262197a1 --- /dev/null +++ b/litellm/router_utils/baseline_request.py @@ -0,0 +1,153 @@ +from __future__ import annotations + +from collections.abc import Iterator, Mapping +from itertools import accumulate +from types import MappingProxyType +from typing import Final, cast + +from pydantic import JsonValue, TypeAdapter, ValidationError + +from litellm.llms.anthropic.pass_through.messages.utils import anthropic_messages_optional_param_keys + +CACHE_SETTINGS: Final = ( + "system", + "instructions", + "tools", + "tool_choice", + "parallel_tool_calls", + "response_format", + "text", + "reasoning", + "reasoning_effort", + "thinking", + "verbosity", + "output_config", + "output_format", + "speed", + "prompt_cache_key", + "cache_key", + "cached_content", + "previous_response_id", + "conversation", + "context_management", + "compaction", +) +_GENERIC_PARAMETERS: Final = ( + *CACHE_SETTINGS, + "prompt_cache_options", + "prompt_cache_retention", + "cache_control", + "max_tokens", + "max_completion_tokens", + "max_output_tokens", + "temperature", + "top_p", + "top_k", + "stop_sequences", + "enable_prompt_caching", + "cache_control_injection_points", + "drop_params", + "additional_drop_params", +) +NATIVE_ONLY_PARAMETERS: Final = tuple( + key + for key in sorted(anthropic_messages_optional_param_keys()) + if key not in (*_GENERIC_PARAMETERS, "metadata", "stream") +) +BASELINE_PARAMETERS: Final = (*_GENERIC_PARAMETERS, *NATIVE_ONLY_PARAMETERS) +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +_MAX_BYTES: Final = 4 * 1024 * 1024 +_MAX_NODES: Final = 32768 +_MAX_DEPTH: Final = 32 + + +def _json_cost(value: object, depth: int = 0) -> Iterator[int]: + if depth > _MAX_DEPTH: + yield _MAX_BYTES + 1 + elif isinstance(value, str): + yield (6 if value.isascii() else 12) * len(value) + 2 + elif isinstance(value, dict): + yield 2 + for key, item in cast(dict[object, object], value).items(): + yield from _json_cost(key, depth + 1) + yield from _json_cost(item, depth + 1) + yield 2 + elif isinstance(value, (list, tuple)): + yield 2 + for item in cast(list[object] | tuple[object, ...], value): + yield from _json_cost(item, depth + 1) + yield 1 + elif isinstance(value, int) and value.bit_length() > 64: + yield _MAX_BYTES + 1 + elif value is None or isinstance(value, (bool, int, float)): + yield 32 + else: + yield _MAX_BYTES + 1 + + +def within_baseline_budget(value: object) -> bool: + return all( + size <= _MAX_BYTES and nodes <= _MAX_NODES for nodes, size in enumerate(accumulate(_json_cost(value)), 1) + ) + + +def _parameters(value: object, *, envelope: bool = False) -> dict[str, object]: + if not isinstance(value, Mapping): + return {} + mapping: Final = cast(Mapping[str, object], value) + keys: Final = (*BASELINE_PARAMETERS, "messages") if envelope else BASELINE_PARAMETERS + return {key: mapping[key] for key in keys if key in mapping} + + +def capture_baseline_parameters( + kwargs: Mapping[str, object], *, include_extra_body: bool = True +) -> Mapping[str, JsonValue] | None: + extra: Final = ( + {"extra_body": _parameters(kwargs.get("extra_body"), envelope=True)} + if include_extra_body and "extra_body" in kwargs + else {} + ) + parameters: Final = {**_parameters(kwargs), **extra} + if not within_baseline_budget(parameters): + return None + try: + return MappingProxyType(_JSON_OBJECT.validate_python(parameters)) + except ValidationError: + return None + + +def baseline_request( + kwargs: Mapping[str, object], + caller: Mapping[str, JsonValue], + deployment: Mapping[str, object], + *, + include_extra_body: bool = True, +) -> Mapping[str, object] | None: + snapshot: Final = capture_baseline_parameters(deployment) + if snapshot is None: + return None + configured: Final = { + **_parameters(snapshot), + **(_parameters(snapshot.get("extra_body")) if include_extra_body else {}), + } + requested: Final = {**_parameters(caller), **(_parameters(caller.get("extra_body")) if include_extra_body else {})} + configured_tools: Final = configured.get("tools") or [] + caller_tools: Final = requested.get("tools") or [] + merged_tools: Final = ( + {"tools": [*configured_tools, *caller_tools]} + if (configured_tools or caller_tools) and isinstance(configured_tools, list) and isinstance(caller_tools, list) + else {} + ) + return MappingProxyType( + { + **{key: value for key, value in kwargs.items() if key not in (*BASELINE_PARAMETERS, "extra_body")}, + **configured, + **requested, + **merged_tools, + **( + {"extra_body": caller.get("extra_body", snapshot.get("extra_body"))} + if not include_extra_body and ("extra_body" in caller or "extra_body" in snapshot) + else {} + ), + } + ) diff --git a/litellm/types/router.py b/litellm/types/router.py index 66a5b3540f9..a66c4571b39 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -5,7 +5,7 @@ litellm.Router Types - includes RouterConfig, UpdateRouterConfig, ModelInfo etc import datetime import enum from collections.abc import Container, Mapping, Sequence -from dataclasses import dataclass +from dataclasses import dataclass, field from typing import ( TYPE_CHECKING, Annotated, @@ -21,7 +21,7 @@ from typing import ( from zoneinfo import ZoneInfo, ZoneInfoNotFoundError import httpx -from pydantic import ConfigDict, Field, field_validator, model_validator +from pydantic import ConfigDict, Field, JsonValue, field_validator, model_validator from typing_extensions import Protocol, ReadOnly, Required, TypedDict, runtime_checkable from litellm._logging import verbose_logger @@ -1223,6 +1223,7 @@ class BaselineRouteStamp: router_name: str baseline_model: str baseline_deployment_id: str + request_parameters: Mapping[str, JsonValue] | None = field(default=None, repr=False) @dataclass(frozen=True, slots=True) diff --git a/tests/code_coverage_tests/recursive_detector.py b/tests/code_coverage_tests/recursive_detector.py index 0c53137cfab..7166891bb1a 100644 --- a/tests/code_coverage_tests/recursive_detector.py +++ b/tests/code_coverage_tests/recursive_detector.py @@ -2,6 +2,7 @@ import ast import os IGNORE_FUNCTIONS = [ + "_json_cost", # bounded at depth 32 and consumed under byte/node limits. "_format_type", "remove_additional_properties", "remove_strict_from_schema", diff --git a/tests/proxy_behavior/spend/test_baseline_accounting.py b/tests/proxy_behavior/spend/test_baseline_accounting.py index dbaf32d579f..4298b85cab4 100644 --- a/tests/proxy_behavior/spend/test_baseline_accounting.py +++ b/tests/proxy_behavior/spend/test_baseline_accounting.py @@ -3,7 +3,8 @@ import json import uuid from collections.abc import AsyncIterator, Callable from contextlib import asynccontextmanager -from datetime import datetime, timezone +from dataclasses import replace +from datetime import datetime, timedelta, timezone from typing import Final import pytest @@ -174,6 +175,10 @@ async def test_commit_ack_loss_and_concurrent_duplicate_delivery_are_idempotent( assert await _store(db, after_commit=True).append(event) == "unavailable" store: Final = _store(db) assert set(await asyncio.gather(*(store.append(event) for _ in range(4)))) == {"recorded"} + duplicate: Final = event.model_copy(update={ + "turn": replace(event.turn, turn_at=event.turn.turn_at + timedelta(seconds=1)) + }) + assert await store.append(duplicate) == "recorded" await _log(db, other) assert await store.append(other) == "recorded" if not attributed: diff --git a/tests/unit/litellm_core_utils/test_logging_worker.py b/tests/unit/litellm_core_utils/test_logging_worker.py index 5d4c9e65d9b..f48495d1062 100644 --- a/tests/unit/litellm_core_utils/test_logging_worker.py +++ b/tests/unit/litellm_core_utils/test_logging_worker.py @@ -6,6 +6,7 @@ import asyncio import contextvars import io import logging +from typing import Final from unittest.mock import AsyncMock, patch import pytest @@ -14,6 +15,61 @@ from litellm.constants import LOGGING_WORKER_AGGRESSIVE_CLEAR_COOLDOWN_SECONDS from litellm.litellm_core_utils.logging_worker import LoggingWorker +@pytest.mark.asyncio +@pytest.mark.parametrize("dispatch", ("worker", "flush", "extracted")) +async def test_optional_work_budget_preserves_callback_context_and_reserves_logging_time(dispatch: str) -> None: + from litellm.litellm_core_utils.logging_worker import optional_callback_budget + + worker: Final = LoggingWorker(timeout=1.0) + identity: Final = contextvars.ContextVar("test_callback_identity", default="outside") + results: Final[asyncio.Queue[tuple[str, float]]] = asyncio.Queue() + + async def callback() -> None: + results.put_nowait((identity.get(), optional_callback_budget(3.0))) + + token: Final = identity.set("request") + worker._ensure_queue() + worker.enqueue(callback()) + identity.reset(token) + try: + if dispatch == "worker": + worker.start() + elif dispatch == "flush": + await worker.flush() + else: + assert worker._queue is not None + await worker._process_single_task(worker._queue.get_nowait()) + restored_identity, budget = await asyncio.wait_for(results.get(), timeout=2) + assert restored_identity == "request" + assert 0 < budget <= worker.timeout / 4 + assert identity.get() == "outside" + assert optional_callback_budget(3.0) == 3.0 + finally: + await worker.stop() + + +def test_exit_flush_bounds_optional_work_and_restores_callers_budget() -> None: + from queue import SimpleQueue + + from litellm.litellm_core_utils.logging_worker import optional_callback_budget + + worker: Final = LoggingWorker(timeout=1.0) + observed: Final[SimpleQueue[float]] = SimpleQueue() + + async def callback() -> None: + observed.put(optional_callback_budget(3.0)) + + async def enqueue() -> None: + worker._ensure_queue() + worker.enqueue(callback()) + + asyncio.run(enqueue()) + worker._flush_on_exit() + assert observed.qsize() == 1 + assert 0 < observed.get_nowait() <= worker.timeout / 4 + assert optional_callback_budget(3.0) == 3.0 + + class _RecordCollector(logging.Handler): """Captures emitted log records so a test can assert on real logging output (level, message args, traceback) instead of patching the logger object.""" @@ -205,7 +261,9 @@ class TestLoggingWorker: asyncio.run(log_on_second_loop()) first_loop.run_until_complete(asyncio.sleep(0.1)) failures = [ - task.exception() for task in first_loop_tasks if task.done() and not task.cancelled() and task.exception() + task.exception() + for task in first_loop_tasks + if task.done() and not task.cancelled() and task.exception() ] finally: first_loop.close() diff --git a/tests/unit/litellm_core_utils/test_redact_messages.py b/tests/unit/litellm_core_utils/test_redact_messages.py index 3bb6b379873..6fbcba170f1 100644 --- a/tests/unit/litellm_core_utils/test_redact_messages.py +++ b/tests/unit/litellm_core_utils/test_redact_messages.py @@ -5,14 +5,26 @@ Covers the proxy flow where headers arrive in litellm_params["metadata"]["header but litellm_params["litellm_metadata"] is None. """ -import asyncio, httpx, importlib, json, os, pytest_asyncio, threading +import asyncio +import importlib +import json +import os +import threading +from collections.abc import AsyncIterator, Mapping +from datetime import datetime +from types import MappingProxyType, SimpleNamespace from typing import Final, Optional, Union -from types import SimpleNamespace +from unittest.mock import patch +import httpx import pytest +import pytest_asyncio +from pydantic import JsonValue import litellm +from litellm.constants import LOGGING_WORKER_MAX_TIME_PER_COROUTINE from litellm.integrations.custom_logger import CustomLogger +from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER from litellm.litellm_core_utils.redact_messages import ( _redact_responses_api_output, perform_redaction, @@ -21,18 +33,14 @@ from litellm.litellm_core_utils.redact_messages import ( should_redact_message_logging, ) from litellm.responses.main import mock_responses_api_response -from collections.abc import AsyncIterator -from datetime import datetime -from litellm.constants import LOGGING_WORKER_MAX_TIME_PER_COROUTINE -from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER -from litellm.types.utils import( +from litellm.types.router import BaselineRouteStamp +from litellm.types.utils import ( ModelResponse, ResponsesAPIResponse, StandardLoggingPayload, TextCompletionResponse, ) from tests._vcr_conftest_common import install_live_call_probe, record_vcr_outcome -from unittest.mock import patch @pytest.fixture(autouse=True) @@ -113,9 +121,7 @@ class TestShouldRedactMessageLogging: def test_enable_redaction_via_header_in_litellm_metadata(self): """Headers inside litellm_metadata (SDK direct call) should work.""" details = _make_model_call_details( - litellm_metadata={ - "headers": {"x-litellm-enable-message-redaction": "true"} - }, + litellm_metadata={"headers": {"x-litellm-enable-message-redaction": "true"}}, ) assert should_redact_message_logging(details) is True @@ -217,21 +223,15 @@ class TestPerformRedaction: redacted = perform_redaction(details, result) - assert details["messages"] == [ - {"role": "user", "content": "redacted-by-litellm"} - ] + assert details["messages"] == [{"role": "user", "content": "redacted-by-litellm"}] assert details["prompt"] == "" assert details["input"] == "" logged_response = details["standard_logging_object"]["response"] assert logged_response["usage"] == {"total_tokens": 1} assert logged_response["output"][0]["text"] == "redacted-by-litellm" - assert logged_response["output"][1]["content"][0]["text"] == ( - "redacted-by-litellm" - ) - assert logged_response["output"][2]["summary"][0]["text"] == ( - "redacted-by-litellm" - ) + assert logged_response["output"][1]["content"][0]["text"] == ("redacted-by-litellm") + assert logged_response["output"][2]["summary"][0]["text"] == ("redacted-by-litellm") assert redacted["usage"] == {"total_tokens": 1} assert redacted["output"][0]["text"] == "redacted-by-litellm" @@ -444,9 +444,7 @@ class TestPerformRedaction: tool_call = redacted.choices[0].message.tool_calls[0] assert tool_call.function.arguments == "redacted-by-litellm" assert tool_call.function.name == "get_weather" - assert result.choices[0].message.tool_calls[0].function.arguments == ( - '{"city": "sensitive-city"}' - ) + assert result.choices[0].message.tool_calls[0].function.arguments == ('{"city": "sensitive-city"}') def test_redacts_tool_call_arguments_on_streaming_response_object(self): """Reproduces the Stream=True path where tool calls arrive as deltas.""" @@ -714,12 +712,8 @@ class TestPerformRedaction: } } ], - "vertex_ai_grounding_metadata": [ - {"webSearchQueries": ["sensitive search term"]} - ], - "vertex_ai_url_context_metadata": [ - {"urlMetadata": [{"retrievedUrl": "https://example.com"}]} - ], + "vertex_ai_grounding_metadata": [{"webSearchQueries": ["sensitive search term"]}], + "vertex_ai_url_context_metadata": [{"urlMetadata": [{"retrievedUrl": "https://example.com"}]}], }, } } @@ -749,9 +743,7 @@ class TestPerformRedaction: "vertex_ai_grounding_metadata", [{"webSearchQueries": ["sensitive search term"]}], ) - response._hidden_params["vertex_ai_grounding_metadata"] = [ - {"webSearchQueries": ["sensitive search term"]} - ] + response._hidden_params["vertex_ai_grounding_metadata"] = [{"webSearchQueries": ["sensitive search term"]}] details = { "stream": True, @@ -772,12 +764,8 @@ class TestPerformRedaction: "metadata": { "hidden_params": { "response_cost": 0.01, - "vertex_ai_grounding_metadata": [ - {"webSearchQueries": ["sensitive search term"]} - ], - "vertex_ai_url_context_metadata": [ - {"urlMetadata": [{"retrievedUrl": "https://example.com"}]} - ], + "vertex_ai_grounding_metadata": [{"webSearchQueries": ["sensitive search term"]}], + "vertex_ai_url_context_metadata": [{"urlMetadata": [{"retrievedUrl": "https://example.com"}]}], "vertex_ai_safety_ratings": [{"category": "HARM"}], "vertex_ai_citation_metadata": [{"citations": ["source"]}], } @@ -797,11 +785,7 @@ class TestPerformRedaction: def test_redact_async_complete_streaming_response(self): """Test that async_complete_streaming_response is properly redacted.""" response_obj = litellm.ModelResponse( - choices=[ - litellm.Choices( - message=litellm.Message(content="secret content", role="assistant") - ) - ] + choices=[litellm.Choices(message=litellm.Message(content="secret content", role="assistant"))] ) model_call_details = { @@ -820,11 +804,7 @@ class TestPerformRedaction: def test_redact_complete_streaming_response(self): """Test that complete_streaming_response is properly redacted.""" response_obj = litellm.ModelResponse( - choices=[ - litellm.Choices( - message=litellm.Message(content="secret content", role="assistant") - ) - ] + choices=[litellm.Choices(message=litellm.Message(content="secret content", role="assistant"))] ) model_call_details = { @@ -842,11 +822,7 @@ class TestPerformRedaction: def test_streaming_responses_untouched_when_disabled(self): response_obj = litellm.ModelResponse( - choices=[ - litellm.Choices( - message=litellm.Message(content="secret content", role="assistant") - ) - ] + choices=[litellm.Choices(message=litellm.Message(content="secret content", role="assistant"))] ) model_call_details = { @@ -909,11 +885,7 @@ class TestPerformRedaction: class TestRedactStreamingResponsesForCustomLogger: def _model_call_details(self): response_obj = litellm.ModelResponse( - choices=[ - litellm.Choices( - message=litellm.Message(content="secret content", role="assistant") - ) - ] + choices=[litellm.Choices(message=litellm.Message(content="secret content", role="assistant"))] ) return { "stream": True, @@ -947,7 +919,10 @@ class TestRedactStreamingResponsesForCustomLogger: @pytest.mark.parametrize("callback_only", [False, True]) def test_classifier_audit_redaction_removes_both_fields_and_source_carrier(callback_only: bool) -> None: - audit: Final = {"classifier_input": {"system": "private rubric"}, "originating_request_masked": {"input": "private source"}} + audit: Final = { + "classifier_input": {"system": "private rubric"}, + "originating_request_masked": {"input": "private source"}, + } standard_payload: Final = { **audit, "messages": [{"role": "user", "content": "private prompt"}], @@ -956,7 +931,9 @@ def test_classifier_audit_redaction_removes_both_fields_and_source_carrier(callb } details: Final = { "standard_logging_object": standard_payload, - "litellm_params": {"proxy_server_request": {"body": {}, "originating_request_masked": audit["originating_request_masked"]}}, + "litellm_params": { + "proxy_server_request": {"body": {}, "originating_request_masked": audit["originating_request_masked"]} + }, } logger: Final = CustomLogger() logger.turn_off_message_logging = True @@ -966,7 +943,10 @@ def test_classifier_audit_redaction_removes_both_fields_and_source_carrier(callb assert "originating_request_masked" not in redacted["standard_logging_object"] assert "originating_request_masked" not in redacted["litellm_params"]["proxy_server_request"] assert details["standard_logging_object"]["classifier_input"] == audit["classifier_input"] - assert details["litellm_params"]["proxy_server_request"]["originating_request_masked"] == audit["originating_request_masked"] + assert ( + details["litellm_params"]["proxy_server_request"]["originating_request_masked"] + == audit["originating_request_masked"] + ) else: perform_redaction(details, result=None) assert "classifier_input" not in details["standard_logging_object"] @@ -981,7 +961,9 @@ def test_classifier_audit_redaction_removes_both_fields_and_source_carrier(callb @pytest.mark.parametrize("excluded", [False, True]) def test_classifier_callback_redaction_preserves_exclusions(monkeypatch: pytest.MonkeyPatch, excluded: bool) -> None: - monkeypatch.setattr(litellm, "standard_logging_payload_excluded_fields", ["messages", "response"] if excluded else []) + monkeypatch.setattr( + litellm, "standard_logging_payload_excluded_fields", ["messages", "response"] if excluded else [] + ) payload: Final = { "classifier_input": {"system": "private rubric"}, "originating_request_masked": {"input": "private source"}, @@ -991,7 +973,9 @@ def test_classifier_callback_redaction_preserves_exclusions(monkeypatch: pytest. } logger: Final = CustomLogger() logger.turn_off_message_logging = True - redacted: Final = logger.redact_standard_logging_payload_from_model_call_details({"standard_logging_object": payload}) + redacted: Final = logger.redact_standard_logging_payload_from_model_call_details( + {"standard_logging_object": payload} + ) stored: Final = redacted["standard_logging_object"] assert "classifier_input" not in stored assert "originating_request_masked" not in stored @@ -1017,16 +1001,18 @@ class _SelfRedactingLogger(CustomLogger): @pytest.mark.parametrize("logger", [CustomLogger(), _SelfRedactingLogger()], ids=["default", "redacts_itself"]) -def test_field_exclusion_alone_leaves_messages_and_responses_intact(monkeypatch: pytest.MonkeyPatch, logger: CustomLogger) -> None: +def test_field_exclusion_alone_leaves_messages_and_responses_intact( + monkeypatch: pytest.MonkeyPatch, logger: CustomLogger +) -> None: monkeypatch.setattr(litellm, "standard_logging_payload_excluded_fields", ["model"]) payload: Final = { "messages": [{"role": "user", "content": "private prompt"}], "response": {"choices": [{"message": {"content": "private answer"}}]}, "model": "classifier", } - stored: Final = logger.redact_standard_logging_payload_from_model_call_details({"standard_logging_object": payload})[ - "standard_logging_object" - ] + stored: Final = logger.redact_standard_logging_payload_from_model_call_details( + {"standard_logging_object": payload} + )["standard_logging_object"] assert stored == {"messages": payload["messages"], "response": payload["response"]} @@ -1038,9 +1024,9 @@ def test_a_callback_that_redacts_itself_keeps_its_messages_but_not_the_classifie } logger: Final = _SelfRedactingLogger() logger.turn_off_message_logging = True - stored: Final = logger.redact_standard_logging_payload_from_model_call_details({"standard_logging_object": payload})[ - "standard_logging_object" - ] + stored: Final = logger.redact_standard_logging_payload_from_model_call_details( + {"standard_logging_object": payload} + )["standard_logging_object"] assert "classifier_input" not in stored assert stored["messages"] == payload["messages"] assert stored["response"] == payload["response"] @@ -1054,19 +1040,54 @@ def test_perform_redaction_drops_the_served_output_texts_from_the_callback_kwarg assert SERVED_OUTPUT_TEXTS_KEY not in details +@pytest.mark.parametrize("callback_only", (False, True)) +@pytest.mark.parametrize("with_standard_payload", (False, True)) +def test_baseline_snapshots_are_redacted_without_mutating_request_state( + callback_only: bool, with_standard_payload: bool +) -> None: + snapshot: Final[Mapping[str, JsonValue]] = MappingProxyType({"system": "private system"}) + route: Final = BaselineRouteStamp("router", "baseline", "deployment", snapshot) + metadata: Final = {"_autorouter_baseline_route": route, "session_id": "session"} + params: Final = {"metadata": metadata, "litellm_metadata": metadata} + details: Final = { + "litellm_params": params, + **({"standard_logging_object": {"model": "model"}} if with_standard_payload else {}), + } + logger: Final = CustomLogger() + logger.turn_off_message_logging = True + if not callback_only: + perform_redaction(details, None) + redacted: Final = ( + logger.redact_standard_logging_payload_from_model_call_details(details) if callback_only else details + ) + expected: Final = BaselineRouteStamp(route.router_name, route.baseline_model, route.baseline_deployment_id) + assert redacted["litellm_params"] == { + key: {"_autorouter_baseline_route": expected, "session_id": "session"} + for key in ("metadata", "litellm_metadata") + } + assert route.request_parameters is snapshot + assert params["metadata"]["_autorouter_baseline_route"] is route + assert params["litellm_metadata"]["_autorouter_baseline_route"] is route + if callback_only: + assert details["litellm_params"] is params + + @pytest.fixture() def _vcr_outcome_gate(request, vcr): install_live_call_probe(request, vcr) yield record_vcr_outcome(request, vcr) + @pytest_asyncio.fixture(loop_scope="function") async def drain_logging_worker(isolate_litellm_state: None) -> AsyncIterator[None]: yield await asyncio.wait_for(GLOBAL_LOGGING_WORKER.flush(), timeout=LOGGING_WORKER_DRAIN_TIMEOUT_SECONDS) + LOGGING_WORKER_DRAIN_TIMEOUT_SECONDS: Final = LOGGING_WORKER_MAX_TIME_PER_COROUTINE + 5.0 + @pytest.fixture(scope="function") def isolate_litellm_state(): """ @@ -1104,6 +1125,7 @@ def isolate_litellm_state(): if attr in _DEFAULTS: setattr(litellm, attr, _DEFAULTS[attr]) + _LIST_ATTRS = ( "callbacks", "success_callback", @@ -1131,6 +1153,7 @@ _SCALAR_ATTRS = ( _DEFAULTS: dict = {} + @pytest.fixture(scope="module") def setup_and_teardown(): """ @@ -1153,6 +1176,7 @@ def setup_and_teardown(): litellm.in_memory_llm_clients_cache.flush_cache() yield + class TestCustomLogger(CustomLogger): def __init__(self, *args, **kwargs): super().__init__(*args, **kwargs) @@ -1164,6 +1188,7 @@ class TestCustomLogger(CustomLogger): self.logged_standard_logging_payload = standard_logging_payload self.response_obj = response_obj + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_global_redaction_on(): @@ -1187,6 +1212,7 @@ async def test_global_redaction_on(): json.dumps(standard_logging_payload, indent=2), ) + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.parametrize( "dynamic_turn_off, expect_redacted", @@ -1213,6 +1239,7 @@ async def test_dynamic_turn_off_message_logging_overrides_global_on(dynamic_turn assert standard_logging_payload["response"]["choices"][0]["message"]["content"] == expected_response_content assert standard_logging_payload["messages"][0]["content"] == expected_message_content + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.parametrize( "dynamic_turn_off, expect_redacted", @@ -1239,6 +1266,7 @@ async def test_dynamic_turn_off_message_logging_overrides_global_off(dynamic_tur assert standard_logging_payload["response"]["choices"][0]["message"]["content"] == expected_response_content assert standard_logging_payload["messages"][0]["content"] == expected_message_content + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_redaction_with_custom_logger_streaming(): @@ -1284,6 +1312,7 @@ async def test_redaction_with_custom_logger_streaming(): finally: litellm.turn_off_message_logging = False + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_streaming_redaction_scoped_to_opted_out_logger(): @@ -1311,6 +1340,7 @@ async def test_streaming_redaction_scoped_to_opted_out_logger(): finally: litellm.callbacks = [] + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_redaction_responses_api(): @@ -1355,6 +1385,7 @@ async def test_redaction_responses_api(): json.dumps(standard_logging_payload, indent=2), ) + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_redaction_responses_api_stream(): @@ -1430,6 +1461,7 @@ async def test_redaction_responses_api_stream(): json.dumps(standard_logging_payload, indent=2), ) + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_redaction_responses_api_with_reasoning_summary(): @@ -1490,6 +1522,7 @@ async def test_redaction_responses_api_with_reasoning_summary(): assert model_call_details["messages"][0]["content"] == "redacted-by-litellm", "Input messages should be redacted" + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_redaction_with_coroutine_objects(): @@ -1535,6 +1568,7 @@ async def test_redaction_with_coroutine_objects(): result = perform_redaction({}, mock_iter) assert result == {"text": "redacted-by-litellm"} + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_redaction_with_streaming_response(): @@ -1570,6 +1604,7 @@ async def test_redaction_with_streaming_response(): json.dumps(standard_logging_payload, indent=2), ) + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_disable_redaction_header_responses_api(): @@ -1605,6 +1640,7 @@ async def test_disable_redaction_header_responses_api(): assert response["output"][0]["content"][0]["text"] == "This is a test response" assert standard_logging_payload["messages"][0]["content"] == "hi" + @pytest.mark.usefixtures("_vcr_outcome_gate", "drain_logging_worker", "isolate_litellm_state", "setup_and_teardown") @pytest.mark.asyncio async def test_redaction_with_metadata_completion_api(): diff --git a/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index c3d4dba7376..ac55e8e9350 100644 --- a/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/unit/llms/anthropic/pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -684,12 +684,12 @@ def _empty_block_msgs(): def test_handler_strips_when_no_presanitized_flag(): """Sync entry point (no async wrapper): handler must still sanitize.""" - from litellm.llms.anthropic.pass_through.messages import handler + from litellm.llms.anthropic.pass_through.messages import handler, utils with patch.object( - handler, + utils, "strip_empty_content_blocks_from_anthropic_messages", - wraps=handler.strip_empty_content_blocks_from_anthropic_messages, + wraps=utils.strip_empty_content_blocks_from_anthropic_messages, ) as spy: result = handler.anthropic_messages_handler( max_tokens=10, @@ -704,12 +704,12 @@ def test_handler_strips_when_no_presanitized_flag(): def test_handler_skips_strip_when_presanitized(): """Async wrapper already sanitized -> handler must NOT rescan.""" - from litellm.llms.anthropic.pass_through.messages import handler + from litellm.llms.anthropic.pass_through.messages import handler, utils with patch.object( - handler, + utils, "strip_empty_content_blocks_from_anthropic_messages", - wraps=handler.strip_empty_content_blocks_from_anthropic_messages, + wraps=utils.strip_empty_content_blocks_from_anthropic_messages, ) as spy: result = handler.anthropic_messages_handler( max_tokens=10, @@ -809,7 +809,7 @@ def test_presanitized_flag_not_leaked_to_provider_params(): @pytest.mark.asyncio async def test_async_wrapper_sets_presanitized_and_sanitizes_once(): """End-to-end: wrapper sanitizes (once) AND signals the handler to skip.""" - from litellm.llms.anthropic.pass_through.messages import handler + from litellm.llms.anthropic.pass_through.messages import handler, utils captured = {} @@ -825,9 +825,9 @@ async def test_async_wrapper_sets_presanitized_and_sanitizes_once(): patch.object(handler, "anthropic_messages_handler", side_effect=fake_handler), patch("asyncio.get_event_loop", return_value=fake_loop), patch.object( - handler, + utils, "strip_empty_content_blocks_from_anthropic_messages", - wraps=handler.strip_empty_content_blocks_from_anthropic_messages, + wraps=utils.strip_empty_content_blocks_from_anthropic_messages, ) as spy, ): await handler.anthropic_messages( diff --git a/tests/unit/llms/anthropic/pass_through/messages/test_request_optional_param_utils.py b/tests/unit/llms/anthropic/pass_through/messages/test_request_optional_param_utils.py index dd744ca66a1..7a4f3aa5f90 100644 --- a/tests/unit/llms/anthropic/pass_through/messages/test_request_optional_param_utils.py +++ b/tests/unit/llms/anthropic/pass_through/messages/test_request_optional_param_utils.py @@ -11,7 +11,7 @@ import pytest import litellm from litellm.llms.anthropic.pass_through.messages.utils import ( AnthropicMessagesRequestUtils, - _anthropic_messages_optional_param_keys, + anthropic_messages_optional_param_keys, ) @@ -30,16 +30,16 @@ def test_optional_param_filtering_unchanged(): def test_valid_keys_are_memoized(): - _anthropic_messages_optional_param_keys.cache_clear() - first = _anthropic_messages_optional_param_keys() + anthropic_messages_optional_param_keys.cache_clear() + first = anthropic_messages_optional_param_keys() for _ in range(50): AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param({"temperature": 0.1}) - info = _anthropic_messages_optional_param_keys.cache_info() + info = anthropic_messages_optional_param_keys.cache_info() # Resolved exactly once despite many calls. assert info.misses == 1 assert info.hits >= 50 # Stable identity (frozenset) returned each call. - assert _anthropic_messages_optional_param_keys() is first + assert anthropic_messages_optional_param_keys() is first assert isinstance(first, frozenset) assert "temperature" in first and "tools" in first diff --git a/tests/unit/proxy/hooks/test_autorouter_baseline_cache.py b/tests/unit/proxy/hooks/test_autorouter_baseline_cache.py index 0f4fd2ff5cb..1be02d2f269 100644 --- a/tests/unit/proxy/hooks/test_autorouter_baseline_cache.py +++ b/tests/unit/proxy/hooks/test_autorouter_baseline_cache.py @@ -3,6 +3,7 @@ import json from collections.abc import AsyncIterator, Callable, Generator, Mapping from contextlib import contextmanager from datetime import datetime +from itertools import product from types import MappingProxyType from typing import Final, cast from uuid import uuid4 @@ -42,7 +43,7 @@ _MESSAGES_JSON: Final = """[{"role":"user","content":[ _MODELS: Final = _MESSAGES.validate_json("""[ {"model_name":"test-router","litellm_params":{"model":"auto_router/complexity_router", "complexity_router_config":{"tiers":{"SIMPLE":"sonnet","MEDIUM":"sonnet","COMPLEX":"sonnet", - "REASONING":"opus"},"session_affinity":false, + "REASONING":{"model_name":"opus","litellm_params":{"max_tokens":16}}},"session_affinity":false, "keyword_tier_rules":[{"keywords":["USE_OPUS"],"tier":"REASONING"}]}}}, {"model_name":"sonnet","litellm_params":{"model":"anthropic/claude-sonnet-5","api_key":"test-selected"}, "model_info":{"id":"selected"}}, @@ -82,7 +83,9 @@ class _CallContext(TypedDict): def _kwargs(logging_obj: Logging, trusted: bool = True, *, explicit_logging: bool = True) -> _CallContext: - context: Final = _OBJECTS.validate_json('{"litellm_metadata":{"user_api_key_hash":"test-caller-hash"}}') + context: Final = _OBJECTS.validate_json( + '{"max_tokens":16,"litellm_metadata":{"user_api_key_hash":"test-caller-hash"}}' + ) Router._record_routing_decision( # pyright: ignore[reportUnknownMemberType, reportPrivateUsage] # production trusted stamp owner context, StandardLoggingRoutingDecision( @@ -133,7 +136,10 @@ def _upstream(request: httpx.Request) -> httpx.Response: assert isinstance(model, str) stream: Final = body.get("stream") is True content: Final = b"".join(_sse(model=model)) if stream else json.dumps(_message(True, model)).encode() - return httpx.Response(200, content=content, request=request, + return httpx.Response( + 200, + content=content, + request=request, headers=MappingProxyType({"content-type": "text/event-stream" if stream else "application/json"}), ) @@ -182,6 +188,7 @@ async def _call( stream: Final = cast(AsyncIterator[object], response) # cast-ok: iterator checked; all items satisfy object assert tuple([chunk async for chunk in stream]) + class _Capture(CustomLogger): def __init__(self, call_id: str) -> None: self.call_id: Final = call_id @@ -199,9 +206,20 @@ class _Capture(CustomLogger): class _Rig: - def __init__(self, monkeypatch: pytest.MonkeyPatch, *, retries: int = 0, count: TokenCounter = _count) -> None: - self.router: Final = Router(model_list=_MODELS, num_retries=retries, - retry_policy=RetryPolicy(RateLimitErrorRetries=retries), disable_cooldowns=True) + def __init__( + self, + monkeypatch: pytest.MonkeyPatch, + *, + retries: int = 0, + count: TokenCounter = _count, + models: list[dict[str, JsonValue]] = _MODELS, + ) -> None: + self.router: Final = Router( + model_list=models, + num_retries=retries, + retry_policy=RetryPolicy(RateLimitErrorRetries=retries), + disable_cooldowns=True, + ) def router() -> Router: return self.router @@ -218,9 +236,16 @@ class _Rig: monkeypatch.setattr(litellm, "_async_success_callback", [self.capture]) def logging(self, stream: bool = False) -> Logging: - return Logging(model="anthropic/claude-sonnet-5", messages=_MESSAGES.validate_json(_MESSAGES_JSON), - stream=stream, call_type=CallTypes.anthropic_messages.value, start_time=datetime.now(), - litellm_call_id=self.call_id, function_id=self.call_id, kwargs={"litellm_session_id":"baseline-session"}) + return Logging( + model="anthropic/claude-sonnet-5", + messages=_MESSAGES.validate_json(_MESSAGES_JSON), + stream=stream, + call_type=CallTypes.anthropic_messages.value, + start_time=datetime.now(), + litellm_call_id=self.call_id, + function_id=self.call_id, + kwargs={"litellm_session_id": "baseline-session"}, + ) def _observation(payload: Mapping[str, object]) -> CapturedBaselineObservation: @@ -232,7 +257,9 @@ def _observation(payload: Mapping[str, object]) -> CapturedBaselineObservation: @pytest.mark.parametrize("stream,baseline", ((False, False), (True, False), (False, True), (True, True))) async def test_native_logging_captures_usage_without_publishing_hypothetical_savings( - monkeypatch: pytest.MonkeyPatch, stream: bool, baseline: bool, + monkeypatch: pytest.MonkeyPatch, + stream: bool, + baseline: bool, ) -> None: rig: Final = _Rig(monkeypatch) messages: Final = _MESSAGES_JSON.replace("question", "question USE_OPUS") if baseline else _MESSAGES_JSON @@ -283,11 +310,14 @@ async def test_caller_cannot_forge_an_observation_scope(monkeypatch: pytest.Monk assert payload["autorouter_savings"] is None -@pytest.mark.parametrize("model,key,endpoint", ( - ("claude-sonnet-5", "test-first", None), - ("claude-opus-5", "test-second", None), - ("claude-opus-5", "test-first", "https://example.test"), -)) +@pytest.mark.parametrize( + "model,key,endpoint", + ( + ("claude-sonnet-5", "test-first", None), + ("claude-opus-5", "test-second", None), + ("claude-opus-5", "test-first", "https://example.test"), + ), +) async def test_count_memo_is_scoped_to_provider_recipient(model: str, key: str, endpoint: str | None) -> None: counts: Final = iter((5000, 6000)) @@ -304,7 +334,8 @@ async def test_count_memo_is_scoped_to_provider_recipient(model: str, key: str, @pytest.mark.parametrize("stream", (False, True)) async def test_provider_counting_does_not_hold_the_inference_response( - monkeypatch: pytest.MonkeyPatch, stream: bool, + monkeypatch: pytest.MonkeyPatch, + stream: bool, ) -> None: counting: Final = asyncio.Event() release: Final = asyncio.Event() @@ -324,3 +355,744 @@ async def test_provider_counting_does_not_hold_the_inference_response( assert _observation(await rig.capture.payload()).observation.plan is not None finally: release.set() + + +@pytest.mark.parametrize("baseline_effort", (None, "medium")) +@pytest.mark.parametrize( + "automatic_system, caching", + ( + (None, "explicit"), + ("stable system", "request"), + ([{"type": "text", "text": "stable system"}], "request"), + ("stable system", "global"), + ([{"type": "text", "text": "stable system"}], "configured"), + ), +) +async def test_native_tier_switch_uses_baseline_settings_and_preserves_history( + monkeypatch: pytest.MonkeyPatch, + baseline_effort: str | None, + automatic_system: str | list[dict[str, str]] | None, + caching: str, +) -> None: + from litellm.proxy.spend_tracking.baseline_accounting import BaselineHistory, advance_baseline_history + + models: Final = _MESSAGES.validate_python( + [ + { + "model_name": "test-router", + "litellm_params": { + "model": "auto_router/complexity_router", + "complexity_router_config": { + "tiers": { + "SIMPLE": {"model_name": "sonnet", "litellm_params": {"reasoning_effort": "low"}}, + "MEDIUM": {"model_name": "sonnet", "litellm_params": {"reasoning_effort": "low"}}, + "COMPLEX": {"model_name": "sonnet", "litellm_params": {"reasoning_effort": "high"}}, + "REASONING": "opus", + }, + "session_affinity": False, + "keyword_tier_rules": [{"keywords": ["ESCALATE"], "tier": "COMPLEX"}], + }, + }, + }, + _MODELS[1], + { + "model_name": "opus", + "model_info": {"id": "baseline"}, + "litellm_params": { + "model": "anthropic/claude-opus-5", + "api_key": "test-selected", + **({"reasoning_effort": baseline_effort} if baseline_effort else {}), + }, + }, + ] + ) + rig: Final = _Rig(monkeypatch, models=models) + monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", caching == "global") + controls: Final = ( + { + "cache_control_injection_points": [ + {"location": "message", "role": "system", "control": {"type": "ephemeral", "ttl": "1h"}}, + {"location": "message", "index": -1, "control": {"type": "ephemeral", "ttl": "1h"}}, + ] + } + if caching == "configured" + else {"enable_prompt_caching": caching == "request"} + ) + captures: Final[asyncio.Queue[CapturedBaselineObservation]] = asyncio.Queue() + with _transport(_upstream) as route: + for suffix in ("", " ESCALATE"): + log: Final = rig.logging() + await rig.router.anthropic_messages( + model="test-router", + max_tokens=4096, + messages=( + [{"role": "user", "content": "question" + suffix}] + if automatic_system is not None + else _MESSAGES.validate_json(_MESSAGES_JSON.replace("question", "question" + suffix)) + ), + system=automatic_system, + **controls, + litellm_logging_obj=log, + litellm_call_id=rig.call_id, + litellm_metadata={"user_api_key_hash": "test-caller-hash"}, + litellm_session_id="native-tiers", + ) + captures.put_nowait(_observation(await rig.capture.payload())) + first_wire, second_wire = (_JSON_OBJECT.validate_json(call.request.content) for call in route.calls) + assert (first_wire.get("thinking"), first_wire.get("output_config")) != ( + second_wire.get("thinking"), + second_wire.get("output_config"), + ) + first, second = (captures.get_nowait() for _ in range(2)) + assert first.scope == second.scope + assert first.observation.plan is not None and second.observation.plan is not None + assert first.observation.plan.breakpoints[0] == second.observation.plan.breakpoints[0] + assert len(first.observation.plan.breakpoints) == (2 if automatic_system is not None else 1) + history, _ = advance_baseline_history( + BaselineHistory(first_at=0.0), + (first.observation.model_copy(update={"request_id": "first", "started_at": 10000.0, "available_at": 10001.0}),), + ) + _, result = advance_baseline_history( + history, + ( + second.observation.model_copy( + update={"request_id": "second", "started_at": 10020.0, "available_at": 10021.0} + ), + ), + ) + assert result[0].usage is not None and result[0].usage.prompt_tokens_details.cached_tokens == 5000 + + +@pytest.mark.parametrize("call_type", (CallTypes.acompletion, CallTypes.aresponses, CallTypes.anthropic_messages)) +async def test_plain_requests_do_not_initialize_or_warn( + monkeypatch: pytest.MonkeyPatch, + caplog: pytest.LogCaptureFixture, + call_type: CallTypes, +) -> None: + rig: Final = _Rig(monkeypatch) + logging: Final = rig.logging() + await rig.hook.async_pre_call_deployment_hook( + { + "litellm_logging_obj": logging, + "litellm_metadata": {"session_id": "ordinary"}, + }, + call_type, + ) + assert logging.baseline_cache_context is None + assert "baseline observation could not be initialized" not in caplog.text + assert not rig.hook.counts + + +async def test_plain_fallback_invalidates_existing_autorouter_capture(monkeypatch: pytest.MonkeyPatch) -> None: + rig: Final = _Rig(monkeypatch) + logging: Final = rig.logging() + await rig.hook.async_pre_call_deployment_hook(_kwargs(logging), CallTypes.anthropic_messages) + assert logging.baseline_cache_context is not None + await rig.hook.async_pre_call_deployment_hook({"litellm_logging_obj": logging}, CallTypes.anthropic_messages) + assert logging.baseline_observation is not None + assert logging.baseline_observation.observation.reason == "retried_request" + + +async def test_native_count_finishing_after_quarter_worker_budget_keeps_plan_and_spend( + monkeypatch: pytest.MonkeyPatch, +) -> None: + from litellm.litellm_core_utils import logging_worker + from litellm.litellm_core_utils.logging_worker import LoggingWorker + + release: Final = asyncio.Event() + + async def count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int: + if not release.is_set(): + asyncio.get_running_loop().call_later(2.3, release.set) + await release.wait() + return await _count(model, api_key, body) + + worker: Final = LoggingWorker(timeout=8.0) + monkeypatch.setattr(logging_worker, "GLOBAL_LOGGING_WORKER", worker) + rig: Final = _Rig(monkeypatch, count=count) + try: + with _transport(_upstream): + await _call(rig.router, rig.logging()) + payload: Final = await rig.capture.payload() + observed: Final = _observation(payload).observation + assert observed.outcome == "complete" and observed.reason is None + assert observed.plan is not None and observed.plan.breakpoints[0].prefix_tokens == 5000 + assert payload["response_cost"] is not None and worker._timeout_total == 0 + finally: + release.set() + await worker.stop() + + +@pytest.mark.parametrize( + "options,on_deployment", + ( + ({"thinking": {"type": "enabled", "budget_tokens": 2048}}, False), + ({"extra_body": {"speed": "fast", "output_config": {"effort": "high"}}}, False), + *product( + ( + {"container": {"id": "container_test"}}, + {"mcp_servers": [{"type": "url", "name": "test", "url": "https://example.com/mcp"}]}, + {"inference_geo": "us"}, + {"safeguards": [{"type": "default"}]}, + ), + (False, True), + ), + ), +) +async def test_native_baseline_identity_keeps_the_actual_transformed_body( + monkeypatch: pytest.MonkeyPatch, options: dict[str, JsonValue], on_deployment: bool +) -> None: + models: Final = _MESSAGES.validate_python( + [ + *_MODELS[:2], + { + **_MODELS[2], + "litellm_params": { + **_JSON_OBJECT.validate_python(_MODELS[2]["litellm_params"]), + **(options if on_deployment else {}), + }, + }, + ] + ) + rig: Final = _Rig(monkeypatch, models=models) + log: Final = rig.logging() + with _transport(_upstream): + await rig.router.anthropic_messages( + model="test-router", + max_tokens=16, + messages=_MESSAGES.validate_json(_MESSAGES_JSON.replace("question", "question USE_OPUS")), + litellm_logging_obj=log, + litellm_call_id=rig.call_id, + litellm_metadata={"user_api_key_hash": "test-caller-hash"}, + litellm_session_id="native-identical", + **({} if on_deployment else options), + ) + observed: Final = _observation(await rig.capture.payload()).observation + assert observed.outcome == "complete" + assert log.baseline_cache_context is not None + assert observed.baseline_equivalent and observed.usage is not None, ( + log.baseline_cache_context.baseline_body, + log.baseline_cache_context.selected_body_digest, + ) + + from litellm.proxy.spend_tracking.baseline_accounting import BaselineHistory, advance_baseline_history + + _, estimates = advance_baseline_history(BaselineHistory(), (observed,)) + assert estimates[0].provenance == "observed_identical" and estimates[0].usage == observed.usage + + +@pytest.mark.parametrize("tier_limit", (8, 16)) +@pytest.mark.parametrize("extra", ({}, {"max_tokens": 8})) +async def test_native_baseline_identity_respects_caller_limit_and_tier_override( + monkeypatch: pytest.MonkeyPatch, tier_limit: int, extra: dict[str, int] +) -> None: + models: Final = _MESSAGES.validate_python( + [ + { + "model_name": "test-router", + "litellm_params": { + "model": "auto_router/complexity_router", + "complexity_router_config": { + "tiers": { + "SIMPLE": {"model_name": "opus", "litellm_params": {"max_tokens": tier_limit}}, + "MEDIUM": {"model_name": "opus", "litellm_params": {"max_tokens": tier_limit}}, + "COMPLEX": "opus", + "REASONING": "opus", + }, + "session_affinity": False, + }, + }, + }, + {**_MODELS[2], "litellm_params": {**_MODELS[2]["litellm_params"], "max_tokens": 64}}, + ] + ) + rig: Final = _Rig(monkeypatch, models=models) + with _transport(_upstream) as route: + await rig.router.anthropic_messages( + model="test-router", + max_tokens=8, + messages=_MESSAGES.validate_json(_MESSAGES_JSON), + litellm_logging_obj=rig.logging(), + litellm_call_id=rig.call_id, + litellm_metadata={"user_api_key_hash": "test-caller-hash"}, + litellm_session_id="native-limits", + extra_body=extra, + ) + observed: Final = _observation(await rig.capture.payload()).observation + wire: Final = _JSON_OBJECT.validate_json(route.calls.last.request.content) + assert wire["max_tokens"] == tier_limit + assert observed.baseline_equivalent == (tier_limit == 8) + + +@pytest.mark.parametrize("nested", (False, True)) +async def test_native_baseline_projection_matches_wire_parameter_placement( + monkeypatch: pytest.MonkeyPatch, + nested: bool, +) -> None: + counted: Final[asyncio.Queue[Mapping[str, JsonValue]]] = asyncio.Queue() + + async def count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int: + counted.put_nowait(body) + return await _count(model, api_key, body) + + models: Final = _MESSAGES.validate_python( + [ + { + **entry, + "litellm_params": { + **_JSON_OBJECT.validate_python(entry["litellm_params"]), + "model": "anthropic/claude-opus-5", + }, + } + if entry["model_name"] == "sonnet" + else entry + for entry in _MODELS + ] + ) + rig: Final = _Rig(monkeypatch, count=count, models=models) + settings: Final = {"speed": "standard", "thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}} + with _transport(_upstream) as route: + await rig.router.anthropic_messages( + model="test-router", + max_tokens=4096, + messages=_MESSAGES.validate_json(_MESSAGES_JSON), + litellm_logging_obj=rig.logging(), + litellm_call_id=rig.call_id, + litellm_metadata={"user_api_key_hash": "test-caller-hash"}, + litellm_session_id="native-placement", + **({"extra_body": settings} if nested else settings), + ) + captured: Final = _observation(await rig.capture.payload()) + wire: Final = _JSON_OBJECT.validate_json(route.calls.last.request.content) + assert captured.observation.plan is not None and not captured.observation.baseline_equivalent + projected: Final = counted.get_nowait() + assert {key: projected[key] for key in settings if key in projected} == { + key: wire[key] for key in settings if key in wire + } + assert {key: wire[key] for key in settings if key in wire} == ({} if nested else settings) + + +_PARITY_TOOL: Final = {"name": "custom", "input_schema": {"type": "object"}, "cache_control": {"type": "ephemeral"}} +_PARITY_SYSTEM: Final = [{"type": "text", "text": "stable system", "cache_control": {"type": "ephemeral"}}] +_PARITY_POINTS: Final = [{"location": "message", "role": "system"}, {"location": "message", "index": -1}] + + +@pytest.mark.parametrize( + "caller,selected,baseline,summary", + ( + pytest.param({"extra_body": {"cache_control": {"type": "ephemeral"}}}, {}, {}, False, id="envelope-control"), + pytest.param({"extra_body": {"system": _PARITY_SYSTEM}}, {}, {}, False, id="envelope-system"), + pytest.param( + {"extra_body": {"messages": _MESSAGES.validate_json(_MESSAGES_JSON)}}, {}, {}, False, id="envelope-messages" + ), + pytest.param({}, {"tools": [_PARITY_TOOL]}, {}, False, id="selected-tool-mark"), + pytest.param({}, {}, {"tools": [_PARITY_TOOL]}, False, id="baseline-tool-mark"), + pytest.param({}, {"system": _PARITY_SYSTEM}, {"system": "baseline system"}, False, id="selected-system-mark"), + pytest.param({}, {"system": "selected system"}, {"system": _PARITY_SYSTEM}, False, id="baseline-system-mark"), + pytest.param({"system": None}, {}, {"system": "configured system"}, False, id="null-system"), + pytest.param({"thinking": None}, {}, {"thinking": {"type": "adaptive"}}, False, id="null-thinking"), + pytest.param({"tools": None}, {}, {"tools": [_PARITY_TOOL]}, False, id="null-tools"), + pytest.param({"verbosity": "low", "instructions": "ignored"}, {}, {}, False, id="ignored-native-options"), + pytest.param( + {"messages": _MESSAGES.validate_json(_MESSAGES_JSON.replace("stable", " "))}, + {}, + {}, + False, + id="empty-marked-block", + ), + pytest.param({"thinking": {"type": "adaptive"}}, {}, {}, True, id="reasoning-summary"), + pytest.param( + {"thinking": {"type": "adaptive"}, "additional_drop_params": ["thinking.display"]}, + {}, + {}, + True, + id="drop-nested-option", + ), + pytest.param( + {"cache_control_injection_points": _PARITY_POINTS}, + {"tools": [{**_PARITY_TOOL, "name": f"custom_{index}"} for index in range(4)]}, + {}, + False, + id="configured-cap", + ), + ), +) +async def test_native_baseline_projection_matches_direct_baseline_request( + monkeypatch: pytest.MonkeyPatch, + caller: dict[str, JsonValue], + selected: dict[str, JsonValue], + baseline: dict[str, JsonValue], + summary: bool, +) -> None: + counted: Final[asyncio.Queue[Mapping[str, JsonValue]]] = asyncio.Queue() + + async def count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int: + counted.put_nowait(body) + return await _count(model, api_key, body) + + def upstream(request: httpx.Request) -> httpx.Response: + body: Final = _JSON_OBJECT.validate_json(request.content) + model: Final = body.get("model") + assert isinstance(model, str) + return httpx.Response( + 200, + request=request, + json={ + **_message(True, model), + "usage": {"input_tokens": 6000, "output_tokens": 10}, + }, + ) + + models: Final = _MESSAGES.validate_python( + [ + _MODELS[0], + { + **_MODELS[1], + "litellm_params": {**_JSON_OBJECT.validate_python(_MODELS[1]["litellm_params"]), **selected}, + }, + { + **_MODELS[2], + "litellm_params": {**_JSON_OBJECT.validate_python(_MODELS[2]["litellm_params"]), **baseline}, + }, + ] + ) + rig: Final = _Rig(monkeypatch, models=models, count=count) + monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", False) + monkeypatch.setattr(litellm, "reasoning_auto_summary", summary) + monkeypatch.delenv("LITELLM_REASONING_AUTO_SUMMARY", raising=False) + request: Final = { + "messages": [{"role": "user", "content": "question"}], + **({"system": "stable system"} if "system" not in selected and "system" not in baseline else {}), + "max_tokens": 4096, + "enable_prompt_caching": True, + **caller, + } + with _transport(upstream) as route: + await rig.router.anthropic_messages(model="opus", **_JSON_OBJECT.validate_python(request)) + direct: Final = _JSON_OBJECT.validate_json(route.calls.last.request.content) + await rig.router.anthropic_messages( + model="test-router", + litellm_logging_obj=rig.logging(), + litellm_call_id=rig.call_id, + litellm_metadata={"user_api_key_hash": "test-caller-hash"}, + litellm_session_id="native-projection-parity", + **_JSON_OBJECT.validate_python(request), + ) + captured: Final = _observation(await rig.capture.payload()) + assert captured.observation.plan is not None, captured.observation.reason + projected: Final = counted.get_nowait() + assert {key: value for key, value in projected.items() if key not in ("metadata", "stream")} == { + key: value for key, value in direct.items() if key not in ("metadata", "stream") + } + + +@pytest.mark.parametrize( + "selected,baseline,usage_field,observed_value,multiplier", + ( + ({}, {"speed": "fast"}, "speed", "standard", 3.0), + ({"inference_geo": "us"}, {}, "inference_geo", "us", 1.0), + ), +) +async def test_native_baseline_prices_projected_settings_without_changing_actual_spend( + monkeypatch: pytest.MonkeyPatch, + selected: dict[str, JsonValue], + baseline: dict[str, JsonValue], + usage_field: str, + observed_value: str, + multiplier: float, +) -> None: + from litellm.proxy.spend_tracking.baseline_accounting import BaselineHistory, advance_baseline_history + from litellm.proxy.spend_tracking.savings import baseline_cost_snapshot, price_baseline_comparison + from litellm.types.utils import ModelInfo + + def upstream(request: httpx.Request) -> httpx.Response: + body: Final = _JSON_OBJECT.validate_json(request.content) + model: Final = body.get("model") + assert isinstance(model, str) + assert body.get(usage_field) == selected.get(usage_field) + return httpx.Response( + 200, + request=request, + json={ + **_message(True, model), + "usage": { + "input_tokens": 6000, + "output_tokens": 10, + usage_field: observed_value, + "cache_creation": {"ephemeral_5m_input_tokens": 0, "ephemeral_1h_input_tokens": 0}, + }, + }, + ) + + rig: Final = _Rig( + monkeypatch, + models=_MESSAGES.validate_python( + [ + _MODELS[0], + { + **_MODELS[1], + "litellm_params": { + **_JSON_OBJECT.validate_python(_MODELS[1]["litellm_params"]), + **selected, + }, + }, + { + **_MODELS[2], + "litellm_params": { + **_JSON_OBJECT.validate_python(_MODELS[2]["litellm_params"]), + **baseline, + }, + }, + ] + ), + ) + monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", False) + with _transport(upstream): + response: Final = _JSON_OBJECT.validate_python( + await rig.router.anthropic_messages( + model="test-router", + max_tokens=16, + messages=[{"role": "user", "content": "question"}], + litellm_logging_obj=rig.logging(), + litellm_call_id=rig.call_id, + litellm_metadata={"user_api_key_hash": "test-caller-hash"}, + litellm_session_id="baseline-speed", + ) + ) + payload: Final = await rig.capture.payload() + captured: Final = _observation(payload) + _, estimates = advance_baseline_history(BaselineHistory(), (captured.observation,)) + estimate: Final = estimates[0] + assert captured.prices is not None + prices: Final[ModelInfo] = { + **captured.prices, + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "provider_specific_entry": {"fast": 3.0, "us": 2.0}, + } + actual: Final = payload["response_cost"] + assert isinstance(actual, float) + snapshot: Final = baseline_cost_snapshot( + captured.model, + prices, + actual, + _OBJECTS.validate_python(payload["cost_breakdown"]), + None, + ) + comparison: Final = price_baseline_comparison(snapshot, estimate.usage, estimate.provenance) + assert comparison is not None and snapshot.actual_token_cost is not None, estimate.reason + assert comparison.baseline == pytest.approx( + actual + (6000 * 1e-6 + 10 * 2e-6) * multiplier - snapshot.actual_token_cost + ) + assert comparison.actual == actual + assert _JSON_OBJECT.validate_python(response["usage"])[usage_field] == observed_value + + +async def test_native_request_rewritten_after_capture_preserves_spend_without_guessing_baseline( + monkeypatch: pytest.MonkeyPatch, +) -> None: + class RewriteSystem(CustomLogger): + async def async_pre_call_deployment_hook( + self, kwargs: Mapping[str, object], call_type: CallTypes | None + ) -> dict[str, object]: + return {**kwargs, "system": "hook system"} + + rig: Final = _Rig(monkeypatch) + monkeypatch.setattr(litellm, "callbacks", [rig.hook, RewriteSystem()]) + with _transport(_upstream) as route: + await _call(rig.router, rig.logging()) + payload: Final = await rig.capture.payload() + wire: Final = _JSON_OBJECT.validate_json(route.calls.last.request.content) + observed: Final = _observation(payload).observation + assert wire["system"] == "hook system" + assert observed.reason == "unsupported_request_transformation" and observed.plan is None + assert observed.usage is not None + actual: Final = payload["response_cost"] + assert isinstance(actual, float) and actual > 0 + + +@pytest.mark.parametrize("history", ("long_session", "non_ascii")) +async def test_native_baseline_models_long_and_non_ascii_history(monkeypatch: pytest.MonkeyPatch, history: str) -> None: + rounds: Final = tuple( + message + for index in range(1200) + for message in ( + {"role": "assistant", "content": [{"type": "tool_use", "id": f"t{index}", "name": "Read", "input": {}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": f"t{index}", "content": "ok"}]}, + ) + ) + prefix: Final = ( + [{"role": "user", "content": "start"}, *rounds] + if history == "long_session" + else [{"role": "user", "content": "a" * 400_000 + "é"}, {"role": "assistant", "content": "ok"}] + ) + messages: Final = json.dumps([*prefix, *_MESSAGES.validate_json(_MESSAGES_JSON)]) + rig: Final = _Rig(monkeypatch) + with _transport(_upstream): + await _call(rig.router, rig.logging(), messages=messages) + observed: Final = _observation(await rig.capture.payload()).observation + assert observed.outcome == "complete" and observed.plan is not None + + +async def test_native_baseline_abstains_after_selected_tier_compaction(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.router_strategy.complexity_router.context_compaction import compaction_executor + + monkeypatch.setitem( + litellm.model_cost, + "summary-fixture", + { + "litellm_provider": "anthropic", + "mode": "chat", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "supports_anthropic_compaction": True, + }, + ) + models: Final = _MESSAGES.validate_python( + [ + { + "model_name": "test-router", + "litellm_params": { + "model": "auto_router/complexity_router", + "complexity_router_config": { + "tiers": {"SIMPLE": "sonnet", "MEDIUM": "opus", "COMPLEX": "opus", "REASONING": "opus"}, + "keyword_tier_rules": [{"keywords": ["answer"], "tier": "SIMPLE"}], + "session_affinity": False, + "enable_context_window_escalation": False, + "max_tokens_from_tier_model": False, + "context_compaction": {"model": "compactor", "max_tokens": 512}, + }, + }, + }, + { + "model_name": "sonnet", + "litellm_params": {"model": "anthropic/claude-sonnet-5", "api_key": "test-selected"}, + "model_info": {"id": "selected", "max_input_tokens": 512, "max_output_tokens": 64}, + }, + { + "model_name": "opus", + "litellm_params": {"model": "anthropic/claude-opus-5", "api_key": "test-selected"}, + "model_info": {"id": "baseline", "max_input_tokens": 200000, "max_output_tokens": 4096}, + }, + { + "model_name": "compactor", + "litellm_params": {"model": "anthropic/summary-fixture", "api_key": "test-compactor"}, + "model_info": {"id": "compactor"}, + }, + ] + ) + + async def summarize(protocol: object, request: object, parent_model: object = None) -> Mapping[str, object]: + return { + "stop_reason": "compaction", + "content": [{"type": "compaction", "content": "compacted", "signature": "s"}], + "usage": {"input_tokens": 0, "output_tokens": 0}, + } + + messages: Final = json.dumps( + [ + {"role": "user", "content": "Background detail. " * 300}, + {"role": "assistant", "content": "Recorded"}, + {"role": "user", "content": "Answer briefly"}, + ] + ) + rig: Final = _Rig(monkeypatch, models=models) + token: Final = compaction_executor.set(summarize) + try: + with _transport(_upstream) as route: + await _call(rig.router, rig.logging(), messages=messages) + observed: Final = _observation(await rig.capture.payload()).observation + wire: Final = route.calls.last.request.content.decode() + finally: + compaction_executor.reset(token) + assert "compacted" in wire and "Background detail" not in wire + assert observed.reason == "unsupported_request_transformation" and observed.plan is None + assert observed.usage is not None + + +async def test_selected_tier_cache_markers_do_not_hide_an_unmarked_baseline_plan( + monkeypatch: pytest.MonkeyPatch, +) -> None: + models: Final = _MESSAGES.validate_python( + [ + _MODELS[0], + { + "model_name": "sonnet", + "litellm_params": { + "model": "anthropic/claude-sonnet-5", + "api_key": "test-selected", + "cache_control_injection_points": [{"location": "message", "role": "user", "index": -1}], + }, + "model_info": {"id": "selected"}, + }, + _MODELS[2], + ] + ) + rig: Final = _Rig(monkeypatch, models=models) + with _transport(_upstream) as route: + await _call( + rig.router, + rig.logging(), + messages='[{"role":"user","content":[{"type":"text","text":"stable"},{"type":"text","text":"question"}]}]', + ) + observed: Final = _observation(await rig.capture.payload()).observation + wire: Final = route.calls.last.request.content.decode() + assert "cache_control" in wire + assert observed.outcome == "complete" and not observed.baseline_equivalent + assert observed.reason is None and observed.plan is not None and not observed.plan.breakpoints + + +@pytest.mark.parametrize("recovery", ("retry", "fallback")) +async def test_tier_pins_never_enter_the_caller_snapshot_on_later_routing_passes( + monkeypatch: pytest.MonkeyPatch, recovery: str +) -> None: + pinned: Final = {"model_name": "first", "litellm_params": {"reasoning_effort": "high", "max_tokens": 777}} + models: Final = _MESSAGES.validate_python( + [ + { + "model_name": "test-router", + "litellm_params": { + "model": "auto_router/complexity_router", + "complexity_router_config": { + "tiers": {"SIMPLE": pinned, "MEDIUM": pinned, "COMPLEX": pinned, "REASONING": "opus"}, + "session_affinity": False, + }, + }, + }, + { + "model_name": "fallback-router", + "litellm_params": { + "model": "auto_router/complexity_router", + "complexity_router_config": { + "tiers": {"SIMPLE": "sonnet", "MEDIUM": "sonnet", "COMPLEX": "sonnet", "REASONING": "opus"}, + "session_affinity": False, + }, + }, + }, + { + "model_name": "first", + "litellm_params": { + "model": "anthropic/claude-sonnet-5" if recovery == "retry" else "anthropic/claude-haiku-5", + "api_key": "test-selected", + }, + "model_info": {"id": "first"}, + }, + *_MODELS[1:], + ] + ) + rig: Final = _Rig(monkeypatch, models=models, retries=1 if recovery == "retry" else 0) + rig.router.fallbacks = [{"test-router": ["fallback-router"]}] + + def upstream(request: httpx.Request) -> httpx.Response: + return _upstream(request) if route.call_count else _error(request, 429, "first attempt") + + log: Final = rig.logging() + with _transport(upstream) as route: + await _call(rig.router, log) + await rig.capture.payload() + first_wire: Final = _JSON_OBJECT.validate_json(route.calls[0].request.content) + assert first_wire.get("output_config") == {"effort": "high"} and first_wire.get("max_tokens") == 777 + context: Final = log.baseline_cache_context + assert context is not None and context.baseline_body is not None + assert context.baseline_body.get("max_tokens") == 16 + assert "output_config" not in context.baseline_body and "thinking" not in context.baseline_body diff --git a/tests/unit/proxy/spend_tracking/test_baseline_accounting.py b/tests/unit/proxy/spend_tracking/test_baseline_accounting.py index a188d65502d..368704e7a75 100644 --- a/tests/unit/proxy/spend_tracking/test_baseline_accounting.py +++ b/tests/unit/proxy/spend_tracking/test_baseline_accounting.py @@ -111,7 +111,13 @@ def test_prefix_match_expiry_and_usage_pricing_fields(ttl: int) -> None: assert warm.usage.prompt_tokens_details.cached_tokens == 6000 assert cold.usage.prompt_tokens_details.cached_tokens == 0 assert cold.usage.prompt_tokens_details.cache_creation_tokens == 6000 - unaffected: Final = {"prompt_tokens", "total_tokens", "prompt_tokens_details", "cache_read_input_tokens", "cache_creation_input_tokens"} + unaffected: Final = { + "prompt_tokens", + "total_tokens", + "prompt_tokens_details", + "cache_read_input_tokens", + "cache_creation_input_tokens", + } assert warm.usage.model_dump(exclude=unaffected) == first.usage.model_dump(exclude=unaffected) assert cold.usage.model_dump(exclude=unaffected) == first.usage.model_dump(exclude=unaffected) @@ -124,7 +130,10 @@ def test_growth_lookback_and_mixed_ttl_keep_distinct_read_write_buckets(warm_tai second: Final = _replay(first, _observation("second", 10001.0, plan=grown))[-1] assert second.reason == "history_unavailable" history: Final = BaselineHistory( - first_at=1.0, last_at=10000.0, equivalent=False, uncertain_before=1.0, + first_at=1.0, + last_at=10000.0, + equivalent=False, + uncertain_before=1.0, entries=(CacheEntry("tail:300", "tail", 7000, 300, 10000.0, 10300.0),) if warm_tail else (), ) _, estimates = advance_baseline_history(history, (_observation("mixed", 10001.0, plan=grown),)) @@ -134,8 +143,12 @@ def test_growth_lookback_and_mixed_ttl_keep_distinct_read_write_buckets(warm_tai # Anthropic billing locations: B is the highest 1h breakpoint AFTER the highest hit A. # https://platform.claude.com/docs/en/build-with-claude/prompt-caching#mixing-different-ttls (2026-09-15) assert usage.prompt_tokens_details.cached_tokens == (7000 if warm_tail else 0) - assert usage.prompt_tokens_details.cache_creation_token_details.ephemeral_1h_input_tokens == (0 if warm_tail else 6500) - assert usage.prompt_tokens_details.cache_creation_token_details.ephemeral_5m_input_tokens == (0 if warm_tail else 500) + assert usage.prompt_tokens_details.cache_creation_token_details.ephemeral_1h_input_tokens == ( + 0 if warm_tail else 6500 + ) + assert usage.prompt_tokens_details.cache_creation_token_details.ephemeral_5m_input_tokens == ( + 0 if warm_tail else 500 + ) @pytest.mark.parametrize("change", ["prefix", "ttl", "unavailable", "failed", "response_cache"]) @@ -197,3 +210,62 @@ def test_modeled_read_cannot_recharge_the_original_private_write_count() -> None } input_cost, output_cost = cost_per_token("claude-opus-5", warm.usage, model_info=prices) assert input_cost + output_cost == pytest.approx((200 * 1e-6 + 6000 * 1e-7 + 30 * 2e-6) * 2.0 * 1.1) + + +def test_mixed_lifetime_lookback_preserves_a_compatible_native_hit() -> None: + marker: Final = _marker("prefix", 3600, 6000) + history: Final = BaselineHistory( + first_at=1.0, + last_at=10000.0, + equivalent=False, + uncertain_before=1.0, + entries=(CacheEntry(marker.fingerprint, marker.content_fingerprint, 6000, 3600, 10000.0, 13600.0),), + ) + plan: Final = CountedPromptCachePlan( + 7100, (_marker("grown", 3600, 6500, ("prefix",)), _marker("tail", 300, 7000, ("prefix",))) + ) + _, estimates = advance_baseline_history(history, (_observation("next", 10001.0, plan=plan),)) + usage: Final = estimates[0].usage + assert usage is not None, estimates[0].reason + assert usage.prompt_tokens_details.cached_tokens == 6000 + assert usage.prompt_tokens_details.text_tokens == 100 + assert usage.prompt_tokens_details.cache_creation_token_details == CacheCreationTokenDetails( + ephemeral_5m_input_tokens=500, ephemeral_1h_input_tokens=500 + ) + + +def test_short_lifetime_hit_cannot_seed_an_unpaid_long_lifetime_entry() -> None: + first: Final = _observation("initial", plan=CountedPromptCachePlan(6200, (_marker("5", 3600, 6000),))) + short: Final = _observation("short", 13700.0, plan=CountedPromptCachePlan(6200, (_marker("3", 300, 4600),))) + mixed: Final = _observation( + "mixed", + 13710.0, + plan=CountedPromptCachePlan(6200, (_marker("3", 3600, 4600), _marker("4", 300, 5500, ("3",)))), + ) + later: Final = _observation("later", 14710.0, plan=CountedPromptCachePlan(6200, (_marker("3", 3600, 4600),))) + _, _, upgrade, after_expiry = _replay(first, short, mixed, later) + assert upgrade.usage is not None and after_expiry.usage is not None + assert upgrade.usage.prompt_tokens_details.cached_tokens == 4600 + assert upgrade.usage.prompt_tokens_details.cache_creation_token_details.ephemeral_1h_input_tokens == 0 + assert after_expiry.usage.prompt_tokens_details.cached_tokens == 0 + assert after_expiry.usage.prompt_tokens_details.cache_creation_token_details.ephemeral_1h_input_tokens == 4600 + + +@pytest.mark.parametrize("writes", (0, 50, None)) +def test_cache_creation_split_is_optional_only_without_writes(writes: int | None) -> None: + usage: Final = Usage( + prompt_tokens=100, + completion_tokens=10, + total_tokens=110, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=100 - (writes or 0), + cached_tokens=0, + cache_creation_tokens=writes, + ), + ) + observed: Final = _observation("no-split", usage=usage, plan=CountedPromptCachePlan(100, ())) + restored: Final = BaselineObservation.model_validate_json(observed.model_dump_json()) + estimate: Final = _replay(restored)[0] + assert (estimate.usage is not None) is (writes == 0) + if estimate.usage is not None: + assert estimate.usage.prompt_tokens == usage.prompt_tokens diff --git a/tests/unit/router_utils/test_baseline_request.py b/tests/unit/router_utils/test_baseline_request.py new file mode 100644 index 00000000000..37d21597752 --- /dev/null +++ b/tests/unit/router_utils/test_baseline_request.py @@ -0,0 +1,59 @@ +from typing import Final + +from litellm.router_utils.baseline_request import baseline_request, capture_baseline_parameters + + +def test_baseline_snapshot_owns_nested_caller_settings_and_overrides_routed_settings() -> None: + reasoning: Final = {"effort": "medium"} + snapshot: Final = capture_baseline_parameters({"reasoning": reasoning, "verbosity": "low"}) + assert snapshot is not None + reasoning["effort"] = "high" + projected: Final = baseline_request( + {"messages": [{"role": "user", "content": "hello"}], "reasoning": reasoning, "verbosity": "high"}, + snapshot, + {"verbosity": "medium"}, + ) + assert projected == { + "messages": [{"role": "user", "content": "hello"}], + "reasoning": {"effort": "medium"}, + "verbosity": "low", + } + + +def test_oversized_snapshot_fails_closed_before_json_validation() -> None: + assert capture_baseline_parameters({"output_config": {"format": "x" * 5_000_000}}) is None + + +def test_snapshot_retains_extra_body_settings_but_no_credentials() -> None: + assert capture_baseline_parameters({"api_key": "private", "extra_body": {"verbosity": "low"}}) == { + "extra_body": {"verbosity": "low"} + } + + +def test_chat_projection_applies_extra_body_after_top_level_parameters() -> None: + snapshot: Final = capture_baseline_parameters({"verbosity": "high", "extra_body": {"verbosity": "low"}}) + assert snapshot is not None + assert baseline_request({}, snapshot, {}) == {"verbosity": "low"} + + +def test_baseline_projection_keeps_caller_tools_and_request_parameter_precedence() -> None: + from litellm.router import Router + + deployment: Final = { + "tools": [{"type": "function", "function": {"name": "configured"}}], + "tool_choice": "required", + "max_tokens": 64, + } + caller: Final = { + "tools": [{"type": "function", "function": {"name": "caller"}}], + "tool_choice": "auto", + "max_tokens": 128, + } + actual_request: Final = dict(caller) + Router._merge_tools_from_deployment({"litellm_params": deployment}, actual_request) + snapshot: Final = capture_baseline_parameters(caller) + assert snapshot is not None + assert baseline_request({"tools": [{"name": "routed-only"}], "max_tokens": 4}, snapshot, deployment) == { + **deployment, + **actual_request, + } From 740d0435a806eec44e7aa359a9c633a5f9d3531b Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:09:56 -0700 Subject: [PATCH 10/13] test(integration): scope the team-scoped models upstream check to its own model (#45106) Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- .../authorization/test_team_scoped_models.py | 19 ++++++++++++++----- 1 file changed, 14 insertions(+), 5 deletions(-) diff --git a/tests/integration/authorization/test_team_scoped_models.py b/tests/integration/authorization/test_team_scoped_models.py index aea6b9e4ba4..1323c1ee004 100644 --- a/tests/integration/authorization/test_team_scoped_models.py +++ b/tests/integration/authorization/test_team_scoped_models.py @@ -15,10 +15,16 @@ def upstream(gateway: Gateway) -> Iterator[httpx.Client]: yield client -def _observed_models(upstream: httpx.Client) -> list[JsonValue]: +def _observed_requests(upstream: httpx.Client) -> list[JsonValue]: observed: Final = upstream.get("/__observations") observed.raise_for_status() - return [request["body"]["model"] for request in observed.json()["requests"]] + requests: Final = object_value(observed.json())["requests"] + assert isinstance(requests, list) + return requests + + +def _calls_to(observed: list[JsonValue], provider_model: str) -> int: + return sum(object_value(object_value(request)["body"]).get("model") == provider_model for request in observed) def _chat(gateway: Gateway, model: str, key: str) -> httpx.Response: @@ -62,7 +68,8 @@ def test_a_team_model_is_listed_and_served_only_for_keys_of_its_team(gateway: Ga assert served.status_code == 200, served.text refused: Final = _chat(gateway, model, other_key) assert refused.status_code == 400, refused.text - assert _observed_models(upstream) == [provider_model] + observed: Final = _observed_requests(upstream) + assert _calls_to(observed, provider_model) == 1, observed def _v2_team_public_names(gateway: Gateway, key: str, model: str) -> list[JsonValue]: @@ -98,8 +105,10 @@ def test_team_model_alias_routes_a_team_key_to_its_target( response: Final = _chat(gateway, alias, key) assert response.status_code == 200, response.text assert string_value(response.json()["model"]) == alias - assert _observed_models(upstream) == [provider_model] + observed: Final = _observed_requests(upstream) + assert _calls_to(observed, provider_model) == 1, observed unaliased: Final = _chat(gateway, f"alias-{uuid.uuid4().hex}", key) assert unaliased.status_code == 403, unaliased.text assert unaliased.json()["error"]["type"] == "key_model_access_denied" - assert _observed_models(upstream) == [] + after_refusal: Final = _observed_requests(upstream) + assert _calls_to(after_refusal, provider_model) == 0, after_refusal From 2c1847f8a262a1967de3db2dfc9c341286eb71e6 Mon Sep 17 00:00:00 2001 From: ahamedshaik16 Date: Wed, 7 Oct 2026 23:48:45 +0530 Subject: [PATCH 11/13] fix(prometheus): add model_group label to end-to-end latency metrics (#44860) litellm_llm_api_latency_metric, litellm_llm_api_time_to_first_token_metric, litellm_request_total_latency_metric, and litellm_deployment_latency_per_output_token previously carried requested_model/litellm_model_name/model_id but not model_group, so pooled-deployment latency couldn't be grouped by model pool on dashboards -- only the proxy-overhead-only metrics (litellm_overhead_latency_metric and friends) had model_group. All four metrics read enum_values.model_group through the existing prometheus_label_factory plumbing, so no new parameter threading was needed, just the label-list addition. Co-authored-by: ahamedshaik16 <24526479+ahamedshaik16@users.noreply.github.com> --- litellm/types/integrations/prometheus.py | 4 + .../test_prometheus_logging_callbacks.py | 4 + .../integrations/test_prometheus_labels.py | 187 ++++++++++++++++++ 3 files changed, 195 insertions(+) diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index dc552a730ea..a951381a2c9 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -416,6 +416,7 @@ def _resolve_deployment_and_latency_caller_identity_labels( class PrometheusMetricLabels: litellm_llm_api_latency_metric = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.API_KEY_HASH.value, UserAPIKeyLabelNames.API_KEY_ALIAS.value, @@ -430,6 +431,7 @@ class PrometheusMetricLabels: ] litellm_llm_api_time_to_first_token_metric = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.API_KEY_HASH.value, UserAPIKeyLabelNames.API_KEY_ALIAS.value, @@ -444,6 +446,7 @@ class PrometheusMetricLabels: ] litellm_request_total_latency_metric = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.END_USER.value, UserAPIKeyLabelNames.API_KEY_HASH.value, UserAPIKeyLabelNames.API_KEY_ALIAS.value, @@ -516,6 +519,7 @@ class PrometheusMetricLabels: ] litellm_deployment_latency_per_output_token = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.v2_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.MODEL_ID.value, UserAPIKeyLabelNames.API_BASE.value, diff --git a/tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py b/tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py index f1c80bb11ea..78545d3fb62 100644 --- a/tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py +++ b/tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py @@ -433,6 +433,7 @@ def test_set_latency_metrics(prometheus_logger): team_alias="test_team_alias", org_id=None, org_alias=None, + model_group="openai-gpt", requested_model="openai-gpt", model="gpt-5-mini", model_id="model-123", @@ -453,6 +454,7 @@ def test_set_latency_metrics(prometheus_logger): team_alias="test_team_alias", org_id=None, org_alias=None, + model_group="openai-gpt", requested_model="openai-gpt", model="gpt-5-mini", model_id="model-123", @@ -473,6 +475,7 @@ def test_set_latency_metrics(prometheus_logger): team_alias="test_team_alias", org_id=None, org_alias=None, + model_group="openai-gpt", requested_model="openai-gpt", model="gpt-5-mini", model_id="model-123", @@ -1054,6 +1057,7 @@ def test_set_llm_deployment_success_metrics(prometheus_logger): # Verify latency per output token metric prometheus_logger.litellm_deployment_latency_per_output_token.labels.assert_called_once_with( + model_group="my_custom_model_group", litellm_model_name="gpt-5-mini", model_id="model-123", api_base="https://api.openai.com", diff --git a/tests/unit/integrations/test_prometheus_labels.py b/tests/unit/integrations/test_prometheus_labels.py index 8a1f5f5a0a8..ead0ce8ecff 100644 --- a/tests/unit/integrations/test_prometheus_labels.py +++ b/tests/unit/integrations/test_prometheus_labels.py @@ -980,6 +980,193 @@ def test_deployment_tpm_rpm_limit_metrics_emit_model_group_from_enum_values(): _clear_prometheus_registry() +def test_model_group_in_latency_metrics(): + """ + Test that model_group label is present on the end-to-end / per-call + latency metrics needed to build model-group latency dashboards. These + metrics previously only carried requested_model, litellm_model_name and + model_id, none of which identify the model_group a pooled deployment + belongs to -- only the proxy-overhead-only latency metrics + (litellm_overhead_latency_metric and friends) carried model_group. + """ + model_group_label = UserAPIKeyLabelNames.MODEL_GROUP.value + + metrics_with_model_group = [ + "litellm_llm_api_latency_metric", + "litellm_llm_api_time_to_first_token_metric", + "litellm_request_total_latency_metric", + "litellm_deployment_latency_per_output_token", + ] + + for metric_name in metrics_with_model_group: + labels = PrometheusMetricLabels.get_labels(metric_name) + assert ( + model_group_label in labels + ), f"Metric {metric_name} should contain model_group label" + print(f"✅ {metric_name} contains model_group label") + + +def test_model_group_value_flows_through_latency_metrics_label_factory(): + """ + The label being in the allow-list is necessary but not sufficient: the + factory must also carry the value from the enum through to the emitted + label. This would fail if the label were dropped from a metric's list or + if the value plumbing regressed, which the allow-list assertion above + cannot catch on its own. + """ + from unittest.mock import MagicMock + + from litellm.integrations.prometheus import ( + PrometheusLogger, + UserAPIKeyLabelValues, + prometheus_label_factory, + ) + + prometheus_logger = MagicMock() + prometheus_logger._cached_metric_labels = {} + prometheus_logger.label_filters = {} + prometheus_logger.get_labels_for_metric = ( + PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger) + ) + + enum_values = UserAPIKeyLabelValues( + model_group="example-model-group", + litellm_model_name="gpt-4o-mini", + requested_model="example-model-group", + status_code="200", + ) + + for metric_name in [ + "litellm_llm_api_latency_metric", + "litellm_llm_api_time_to_first_token_metric", + "litellm_request_total_latency_metric", + "litellm_deployment_latency_per_output_token", + ]: + labels = prometheus_label_factory( + supported_enum_labels=prometheus_logger.get_labels_for_metric( + metric_name=metric_name + ), + enum_values=enum_values, + ) + assert ( + labels.get("model_group") == "example-model-group" + ), f"{metric_name} should emit model_group=example-model-group, got {labels.get('model_group')!r}" + + +def test_latency_metrics_emit_model_group_from_set_latency_metrics(): + """ + End-to-end emit wiring for _set_latency_metrics. + + The label-list and factory tests above prove the label exists and that + the factory carries a value handed to it, but neither drives the real + _set_latency_metrics code path, so deleting the production model_group + plumbing there would still pass them. This calls it directly with a + streaming request (so the time-to-first-token branch also fires) and + asserts the real litellm_llm_api_latency_metric, + litellm_llm_api_time_to_first_token_metric and + litellm_request_total_latency_metric Histogram series actually carry it. + """ + import datetime + + from litellm.integrations.prometheus import PrometheusLogger, UserAPIKeyLabelValues + + _clear_prometheus_registry() + try: + logger = PrometheusLogger() + start_time = datetime.datetime(2024, 1, 1, 0, 0, 0) + api_call_start_time = datetime.datetime(2024, 1, 1, 0, 0, 1) + completion_start_time = datetime.datetime(2024, 1, 1, 0, 0, 2) + end_time = datetime.datetime(2024, 1, 1, 0, 0, 3) + + enum_values = UserAPIKeyLabelValues( + model_group="example-model-group", + litellm_model_name="gpt-4o-mini", + requested_model="example-model-group", + status_code="200", + ) + + logger._set_latency_metrics( + kwargs={ + "start_time": start_time, + "end_time": end_time, + "api_call_start_time": api_call_start_time, + "completion_start_time": completion_start_time, + "stream": True, + "litellm_params": {"metadata": {}}, + }, + model="gpt-4o-mini", + user_api_key=None, + user_api_key_alias=None, + user_api_team=None, + user_api_team_alias=None, + enum_values=enum_values, + ) + + for metric in ( + logger.litellm_llm_api_latency_metric, + logger.litellm_llm_api_time_to_first_token_metric, + logger.litellm_request_total_latency_metric, + ): + index = metric._labelnames.index("model_group") + values = {sample_key[index] for sample_key in metric._metrics} + assert values == {"example-model-group"}, ( + f"expected model_group=example-model-group on {metric._name}, got {values}" + ) + finally: + _clear_prometheus_registry() + + +def test_deployment_latency_per_output_token_emits_model_group_from_enum_values(): + """ + End-to-end emit wiring for litellm_deployment_latency_per_output_token. + + Drives set_llm_deployment_success_metrics directly (its only caller) with + output_tokens > 0 so the latency-per-token branch fires, and asserts the + real Histogram series carries model_group; fails if that label-list + addition or the enum_values plumbing is removed. + """ + import datetime + + from litellm.integrations.prometheus import PrometheusLogger, UserAPIKeyLabelValues + + _clear_prometheus_registry() + try: + logger = PrometheusLogger() + start_time = datetime.datetime(2024, 1, 1, 0, 0, 0) + end_time = datetime.datetime(2024, 1, 1, 0, 0, 2) + enum_values = UserAPIKeyLabelValues( + model_group="example-model-group", + litellm_model_name="gpt-4o-mini", + requested_model="example-model-group", + status_code="200", + ) + logger.set_llm_deployment_success_metrics( + request_kwargs={ + "model": "gpt-4o-mini", + "litellm_params": {"metadata": {"model_info": {"id": "model-123"}}}, + "standard_logging_object": { + "model_group": "example-model-group", + "model_id": "model-123", + "api_base": "https://api.openai.com", + "hidden_params": {"additional_headers": None, "litellm_overhead_time_ms": None}, + }, + }, + start_time=start_time, + end_time=end_time, + enum_values=enum_values, + output_tokens=10.0, + ) + + metric = logger.litellm_deployment_latency_per_output_token + index = metric._labelnames.index("model_group") + values = {sample_key[index] for sample_key in metric._metrics} + assert values == {"example-model-group"}, ( + f"expected model_group=example-model-group on {metric._name}, got {values}" + ) + finally: + _clear_prometheus_registry() + + if __name__ == "__main__": test_user_email_in_required_metrics() test_user_email_label_exists() From 0586289817b01610250b3e294db3cd14d24cb819 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:28:14 -0700 Subject: [PATCH 12/13] test(mcp): fix stale bridge-hook, applied-guardrails and pagination-revoke integration tests (#45008) * test(mcp): fix stale bridge-hook, applied-guardrails and pagination-revoke integration tests * test(mcp): clear the spare direct grant when restoring the access-group policy --------- Co-authored-by: yuneng --- .../mcp/test_mcp_accounting_guardrails.py | 2 +- tests/integration/mcp/test_mcp_llm_endpoints.py | 4 ++-- tests/integration/mcp/test_pagination.py | 15 ++++++++------- 3 files changed, 11 insertions(+), 10 deletions(-) diff --git a/tests/integration/mcp/test_mcp_accounting_guardrails.py b/tests/integration/mcp/test_mcp_accounting_guardrails.py index a276718c280..0d1c707d1ff 100644 --- a/tests/integration/mcp/test_mcp_accounting_guardrails.py +++ b/tests/integration/mcp/test_mcp_accounting_guardrails.py @@ -468,7 +468,7 @@ def test_generic_sink_and_native_hooks_receive_listed_metadata_on_typed_keys_wit assert row["status"] == "success", row assert _tool_metadata(row)["name"] == "lookup", row metadata: Final = row["metadata"] - assert isinstance(metadata, dict) and metadata["applied_guardrails"] == [hooks_rig.guardrail], metadata + assert isinstance(metadata, dict) and metadata["applied_guardrails"].count(hooks_rig.guardrail) == 1, metadata def test_pre_call_mask_reaches_the_peer_and_post_call_mask_reaches_the_caller_on_one_call_id( diff --git a/tests/integration/mcp/test_mcp_llm_endpoints.py b/tests/integration/mcp/test_mcp_llm_endpoints.py index af3bff1fd8b..9e946dfc253 100644 --- a/tests/integration/mcp/test_mcp_llm_endpoints.py +++ b/tests/integration/mcp/test_mcp_llm_endpoints.py @@ -934,7 +934,7 @@ def test_identical_nonstream_repeat_is_a_cache_hit_without_new_model_peer_or_hoo assert _spend_row(key, repeat.call_id)["cache_hit"] == "True" -def test_messages_bridge_hook_keeps_the_base_shape_without_request_local_metadata(hooked: Hooked) -> None: +def test_messages_bridge_hook_sees_the_definition_the_request_served(hooked: Hooked) -> None: with _bridge_rig(hooked, "messages") as rig: key: Final = _bridge_key(rig) marker: Final = "m" + uuid.uuid4().hex @@ -943,6 +943,6 @@ def test_messages_bridge_hook_keeps_the_base_shape_without_request_local_metadat assert found.status_code == 200, found.text blocked: Final = rig.post(key, probe, [rig.mcp("lookup")]) assert blocked.status_code == 200, blocked.text - assert _echoed(JSON_VALUE.validate_json(blocked.content)) == COLD, blocked.text + assert _echoed(JSON_VALUE.validate_json(blocked.content)) == _served(LOOKUP), blocked.text assert rig.peer_calls() == (("lookup", {"query": marker}),) assert rig.hook_messages(marker) == (f"Tool: lookup\nArguments: {dict(query=marker)}",) diff --git a/tests/integration/mcp/test_pagination.py b/tests/integration/mcp/test_pagination.py index 84689232038..b8b28bd39ce 100644 --- a/tests/integration/mcp/test_pagination.py +++ b/tests/integration/mcp/test_pagination.py @@ -1,7 +1,7 @@ import asyncio from contextlib import asynccontextmanager from pathlib import Path -from typing import Literal +from typing import Final, Literal import httpx import pytest @@ -138,7 +138,7 @@ def test_continuations_reauthorize_and_reject_registry_changes(tmp_path: Path, m assert os.environ.get("DATABASE_URL"), "This integration case requires disposable-database access" - async def exercise(a, b, peer, identity, owner, stranger, policy): + async def exercise(a, b, peer, identity, owner, stranger, policy, spare): owner_a = Gateway(a.client, owner, peer.url) owner_b = Gateway(b.client, owner, peer.url) stranger_b = Gateway(b.client, stranger, peer.url) @@ -165,9 +165,9 @@ def test_continuations_reauthorize_and_reject_registry_changes(tmp_path: Path, m { "key": owner, **( - {"access_group_ids": []} + {"access_group_ids": [], "object_permission": {"mcp_servers": [spare]}} if grant == "access_group" - else {"object_permission": {"mcp_servers": ["no-mcp-servers"]}} + else {"object_permission": {"mcp_servers": [spare]}} ), }, ) @@ -177,7 +177,7 @@ def test_continuations_reauthorize_and_reject_registry_changes(tmp_path: Path, m with pytest.raises(MCPError, match="fresh listing"): await getattr(session, method)(params=PaginatedRequestParams(cursor=first.next_cursor)) assert not any(call["body"].get("method", "").endswith("/list") for call in peer.drain()) - a.post("/key/update", {"key": owner, **policy}) + a.post("/key/update", {"key": owner, "object_permission": {"mcp_servers": []}, **policy}) changed = a.request("PUT", "/v1/mcp/server", {"server_id": identity, "description": "new catalog generation"}) assert changed.status_code == 202, changed.text for method, first in first_pages.items(): @@ -207,7 +207,7 @@ def test_continuations_reauthorize_and_reject_registry_changes(tmp_path: Path, m capture_output=True, text=True, ) - with paginated_mcp_peer() as peer, httpx.Client() as client: + with paginated_mcp_peer() as peer, paginated_mcp_peer() as spare_peer, httpx.Client() as client: seed = Gateway(client, "sk-pagination-test", peer.url) config = tmp_path / "database-proxy.yaml" config.write_text( @@ -231,6 +231,7 @@ def test_continuations_reauthorize_and_reject_registry_changes(tmp_path: Path, m a.scenario() as scenario, ): identity = register_mcp(scenario, peer, "pages") + spare: Final = register_mcp(scenario, spare_peer, "spare") group = a.request( "POST", "/v1/access_group", @@ -248,7 +249,7 @@ def test_continuations_reauthorize_and_reject_registry_changes(tmp_path: Path, m owner = scenario.key(**policy) stranger = scenario.key(object_permission={"mcp_servers": [identity]}) assert owner != stranger - asyncio.run(exercise(a, b, peer, identity, owner, stranger, policy)) + asyncio.run(exercise(a, b, peer, identity, owner, stranger, policy, spare)) @pytest.mark.parametrize("changed", ["key", "snapshot"]) From 31b90ebb47d3b47d40e2c01f416adba2ecb19540 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:29:05 -0700 Subject: [PATCH 13/13] test(observability): wait for warm-up spend rows in capped and count only the post-wipe half in X4 (#45006) Co-authored-by: yuneng --- tests/integration/observability/conftest.py | 8 +++++++- .../test_cache_hit_guardrail_metrics_chaos.py | 6 +++--- 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/tests/integration/observability/conftest.py b/tests/integration/observability/conftest.py index c5150877857..fcac8eaba3c 100644 --- a/tests/integration/observability/conftest.py +++ b/tests/integration/observability/conftest.py @@ -8,8 +8,9 @@ from urllib.parse import urlparse import pytest import yaml +from integration._support.client import eventually from integration._support.otlp_sink import SpanSinks, owned_sinks -from integration._support.prometheus_series import CapRig, series_cap_rig +from integration._support.prometheus_series import CapRig, series_cap_rig, spend_rows from pydantic import JsonValue AuditConfigWriter = Callable[[Path, Mapping[str, JsonValue]], Path] @@ -63,4 +64,9 @@ def capped(tmp_path_factory: pytest.TempPathFactory) -> Iterator[CapRig]: workers=2, warm_keys=3, ) as rig: + eventually( + lambda: tuple(len(spend_rows(key.alias)) for key in rig.warm), + lambda counts: all(count == 1 for count in counts), + seconds=70, + ) yield rig diff --git a/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py b/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py index c77b8eebe33..7e3b479e5f0 100644 --- a/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py +++ b/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py @@ -263,7 +263,7 @@ def test_worker_kill_mid_burst_keeps_counting(gateway: Gateway, tmp_path: Path) def test_proxy_restart_mid_burst_keeps_counting(gateway: Gateway, tmp_path: Path) -> None: - """X4: restart the owned proxy between the two halves; pre-restart count asserted, then recounted.""" + """X4: the boot wipes the kept directory, so the second proxy counts only the second half.""" marker: Final = uuid.uuid4().hex prom_dir: Final = tmp_path / "prom" prom_dir.mkdir() @@ -321,8 +321,8 @@ def test_proxy_restart_mid_burst_keeps_counting(gateway: Gateway, tmp_path: Path _populated(_samples(owned_two.gateway, (model,)), deployment), _blank(_samples(owned_two.gateway, (model,))), ), - lambda observed: observed[0] == len(named) and observed[1] == 0, + lambda observed: observed[0] == len(second_half) and observed[1] == 0, seconds=70, ) - assert post[0] == len(named), (pre, post, outcomes_two) + assert post[0] == len(second_half), (pre, post, outcomes_two) owned_two.gateway.post("/model/delete", {"id": deployment})